outerloop-science 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. outerloop/__init__.py +18 -0
  2. outerloop/__main__.py +3 -0
  3. outerloop/appauth.py +230 -0
  4. outerloop/appmanifest.py +203 -0
  5. outerloop/attempt.py +3784 -0
  6. outerloop/brief.py +528 -0
  7. outerloop/cli.py +621 -0
  8. outerloop/climbboard.py +1395 -0
  9. outerloop/compute.py +654 -0
  10. outerloop/contract.py +492 -0
  11. outerloop/contract_cli.py +63 -0
  12. outerloop/disk.py +164 -0
  13. outerloop/dispatch.py +631 -0
  14. outerloop/evalcache.py +147 -0
  15. outerloop/followup.py +2172 -0
  16. outerloop/github.py +1531 -0
  17. outerloop/harness.py +1435 -0
  18. outerloop/housekeeping.py +151 -0
  19. outerloop/image.py +368 -0
  20. outerloop/init.py +744 -0
  21. outerloop/intake.py +126 -0
  22. outerloop/launchlog.py +239 -0
  23. outerloop/limits.py +80 -0
  24. outerloop/maintain.py +353 -0
  25. outerloop/maintain_agent_cli.py +81 -0
  26. outerloop/maintain_post_cli.py +140 -0
  27. outerloop/markers.py +48 -0
  28. outerloop/measure.py +529 -0
  29. outerloop/orchestrator.py +2011 -0
  30. outerloop/panel.py +188 -0
  31. outerloop/paths.py +40 -0
  32. outerloop/posting.py +160 -0
  33. outerloop/progress.py +170 -0
  34. outerloop/py.typed +0 -0
  35. outerloop/review.py +615 -0
  36. outerloop/review_agent.py +263 -0
  37. outerloop/review_agent_cli.py +209 -0
  38. outerloop/review_post_cli.py +162 -0
  39. outerloop/review_summarize_cli.py +165 -0
  40. outerloop/role_runner.py +229 -0
  41. outerloop/roles.py +274 -0
  42. outerloop/rolespec.py +91 -0
  43. outerloop/runstate.py +385 -0
  44. outerloop/steward.py +845 -0
  45. outerloop/style.py +12 -0
  46. outerloop/syscall.py +1192 -0
  47. outerloop/syscall_cli.py +762 -0
  48. outerloop/tick.py +3422 -0
  49. outerloop/verifier.py +403 -0
  50. outerloop/verify_agent.py +151 -0
  51. outerloop/verify_agent_cli.py +95 -0
  52. outerloop/verify_post_cli.py +116 -0
  53. outerloop/watcher.py +203 -0
  54. outerloop_science-0.1.0.dist-info/METADATA +152 -0
  55. outerloop_science-0.1.0.dist-info/RECORD +59 -0
  56. outerloop_science-0.1.0.dist-info/WHEEL +4 -0
  57. outerloop_science-0.1.0.dist-info/entry_points.txt +2 -0
  58. outerloop_science-0.1.0.dist-info/licenses/LICENSE +202 -0
  59. outerloop_science-0.1.0.dist-info/licenses/NOTICE +5 -0
outerloop/dispatch.py ADDED
@@ -0,0 +1,631 @@
1
+ """The dispatched-eval primitive: an eval as its own Slurm job.
2
+
3
+ This module owns the three pieces docs/design/dispatcher.md specifies,
4
+ and nothing else:
5
+
6
+ * snapshot_tree — a retained snapshot of a dirty workspace, taken
7
+ against a TEMPORARY unique index seeded from the base commit (working
8
+ index untouched, tracked-vs-ignored parity with the drift fingerprint),
9
+ kept reachable under a unique ref so gc cannot prune it before a queued
10
+ job runs. Release with drop_snapshot after the eval is read.
11
+ * write_eval_job — the orchestrator-authored job script: materialize the
12
+ snapshot by CHECKOUT (git worktree add, then delete the .git gitfile so
13
+ the jail sees a plain faithful directory — checkout keeps .gitattributes
14
+ and applies no export processing), run the contract command under the
15
+ SAME jail as the in-job evaluator, capture stdout OUTSIDE the containment
16
+ into the run directory (the jailed process never sees the run dir).
17
+ * read_eval_result — the wake side: exit code + the same last-JSON-line
18
+ metric contract the in-job evaluator parses.
19
+
20
+ Dispatch is chosen per benchmark: `eval_minutes` is a contract HINT with
21
+ its own code-side ceiling; under the in-job threshold nothing here runs.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ import contextlib
27
+ import json
28
+ import logging
29
+ import math
30
+ import re
31
+ import shlex
32
+ import shutil
33
+ import subprocess
34
+ from dataclasses import dataclass
35
+ from pathlib import Path
36
+ from uuid import uuid4
37
+
38
+ from outerloop.compute import JobSpec
39
+ from outerloop.github import (
40
+ SAFE_GIT_FLAGS,
41
+ GitError,
42
+ Workspace,
43
+ ensure_regular_git_dir,
44
+ git_identity,
45
+ )
46
+ from outerloop.orchestrator import EvalError, managed_eval_env, metric_from_output
47
+
48
+ log = logging.getLogger(__name__)
49
+
50
+ # extra_env keys are exported unquoted into the job script; keep them to a
51
+ # shell identifier shape (the values ARE shlex-quoted).
52
+ _SHELL_IDENT = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*$")
53
+
54
+ # In-job evals must fit the climb job's overhead runway (a few minutes each,
55
+ # see limits.ATTEMPT_OVERHEAD_MINUTES); anything longer is dispatched.
56
+ IN_JOB_EVAL_MINUTES = 5
57
+ # Ceiling for the contract's eval_minutes hint: OUR spend cap, not the
58
+ # target's to raise (same grammar as every budget ceiling).
59
+ # A BACKSTOP, not a policy knob: the eval walltime is the contract's hint
60
+ # or the author's own declaration at submit, and spend is metered in
61
+ # GPU-hours against gpu_hours_per_run (syscall.budget_error). This only
62
+ # caps a runaway value.
63
+ EVAL_JOB_MINUTES_CEILING = 1440
64
+ # Slack added to the job walltime beyond the eval itself: worktree
65
+ # materialization + venv build from the lockfile on node-local scratch.
66
+ EVAL_JOB_SETUP_MINUTES = 10
67
+ # a GPU eval trains: data loading + torch.compile workers need real cores
68
+ # and host RAM; sized per GPU so an 8-GPU eval scales the same way
69
+ EVAL_CPUS_PER_GPU = 8
70
+ EVAL_MEM_GB_PER_GPU = 64
71
+
72
+
73
+ def effective_eval_minutes(eval_minutes: int | None) -> int:
74
+ """The contract hint clamped into [1, ceiling]; None means in-job."""
75
+ if eval_minutes is None:
76
+ return 0
77
+ return max(1, min(int(eval_minutes), EVAL_JOB_MINUTES_CEILING))
78
+
79
+
80
+ def should_dispatch(eval_minutes: int | None) -> bool:
81
+ return effective_eval_minutes(eval_minutes) > IN_JOB_EVAL_MINUTES
82
+
83
+
84
+ def afterany_ids(afterany: str) -> list[str]:
85
+ """The job ids inside an ``afterany:<id>:<id>...`` dependency string, or
86
+ ``[]`` when there are none (a blind park carries an empty dependency)."""
87
+ return afterany.split(":")[1:] if afterany else []
88
+
89
+
90
+ @dataclass(frozen=True)
91
+ class Snapshot:
92
+ """A retained snapshot of a dirty workspace. `ref` keeps the commit
93
+ reachable so gc cannot prune it while a queued job still needs it;
94
+ `tree` is the drift fingerprint. Release the ref via `drop_snapshot`
95
+ once the eval has been read."""
96
+
97
+ commit: str
98
+ tree: str
99
+ ref: str
100
+
101
+
102
+ def snapshot_tree(
103
+ ws: Workspace,
104
+ base_sha: str,
105
+ exclude: tuple[str, ...] = (),
106
+ force: tuple[str, ...] = (),
107
+ author: str = "",
108
+ ) -> Snapshot:
109
+ """Snapshot the workspace's current CONTENT as a commit parented on
110
+ `base_sha`, without touching the working index, and retain it under a
111
+ unique ref so gc cannot prune it before a queued job materializes it.
112
+ `exclude` drops those paths (files or whole directories) from the sealed
113
+ tree — research lines use it to keep the agent's memory out of every
114
+ MEASURABLE seal (gate candidates, launches) while the notebook seal
115
+ keeps it (docs/design/research-lines.md). `force` adds those paths even
116
+ when the target's ignore rules match them — the notebook seal uses it so
117
+ a .gitignore entry cannot silently discard session memory; callers pass
118
+ only paths that exist. `author` is the bot login the seal commit is
119
+ made as (empty: OUTERLOOP_BOT_LOGIN).
120
+ """
121
+ # the snapshot writes an index, a tree, a commit, and a ref into this
122
+ # repository: a session-reshaped .git is refused first, like every other
123
+ # kernel git call on a workspace
124
+ ensure_regular_git_dir(Path(ws.root))
125
+ # Unique per snapshot: two snapshots against the SAME base run
126
+ # concurrently (the design's paired baseline/candidate fan-out) and must
127
+ # not collide on one index file.
128
+ token = uuid4().hex
129
+ index = Path(ws.root) / ".git" / f"dispatch-index-{token}"
130
+ ref = f"refs/dispatch/{token}"
131
+ env = {"GIT_INDEX_FILE": str(index)}
132
+ # The workspace is AGENT-WRITTEN: every git call carries SAFE_GIT_FLAGS
133
+ # (hooks/fsmonitor/ssh/credential neutralizers), drops HOME (no
134
+ # ~/.gitconfig read), and disables agent-defined filters via
135
+ # _filter_neutral_env (GIT_CONFIG_* env, which handles driver names with
136
+ # '=' or dots that `-c` cannot). The neutralizers ride EVERY index op
137
+ # because the clean filter runs lazily at write-tree, not add.
138
+ base_git = ["git", "-C", str(ws.root), *SAFE_GIT_FLAGS]
139
+ try:
140
+ # Neutralize agent-defined filters via GIT_CONFIG_KEY_n/VALUE_n env
141
+ # injection, NOT `-c`: a driver name containing '=' (a legal git
142
+ # subsection char) defeats `-c filter.<driver>.clean=cat` because git
143
+ # splits `-c` at the FIRST '='. The env form takes key and value as
144
+ # SEPARATE strings, immune to that — and to dots. See _filter_neutral_env.
145
+ neutral = _filter_neutral_env(base_git, env)
146
+ run_env = {**env, **neutral}
147
+
148
+ def run(args: list[str], timeout: int) -> str:
149
+ return subprocess.run(
150
+ args,
151
+ env=_git_env(run_env),
152
+ check=True,
153
+ capture_output=True,
154
+ text=True,
155
+ timeout=timeout,
156
+ ).stdout.strip()
157
+
158
+ git = base_git # neutralizers now ride the ENV, not argv
159
+
160
+ # seed from the base so ignore rules apply as they do to a populated
161
+ # index (a fresh empty index would drop tracked-but-ignored files)
162
+ run([*git, "read-tree", base_sha], 60)
163
+ run([*git, "add", "-A"], 120)
164
+ if force:
165
+ run([*git, "add", "-f", "--", *force], 60)
166
+ if exclude:
167
+ run([*git, "rm", "--cached", "-r", "-q", "--ignore-unmatch", "--", *exclude], 60)
168
+ # .gitattributes are KEPT: the job materializes the tree by CHECKOUT
169
+ # (git worktree), which reproduces content faithfully — including
170
+ # .gitattributes — and does NOT apply export-ignore/export-subst
171
+ # (those are `git archive` only). So fidelity and integrity hold
172
+ # together; the filter side is already neutralized via GIT_CONFIG env.
173
+ tree = run([*git, "write-tree"], 60)
174
+ commit = run(
175
+ [
176
+ *git,
177
+ *git_identity(author),
178
+ "commit-tree",
179
+ tree,
180
+ "-p",
181
+ base_sha,
182
+ "-m",
183
+ "dispatch snapshot",
184
+ ],
185
+ 60,
186
+ )
187
+ # retain: an unreachable commit-tree object can be pruned by gc while
188
+ # the eval job is still queued
189
+ run([*base_git, "update-ref", ref, commit], 30)
190
+ return Snapshot(commit=commit, tree=tree, ref=ref)
191
+ except subprocess.CalledProcessError as exc:
192
+ raise EvalError(f"snapshot failed: {exc.stderr.strip()[:300]}") from exc
193
+ except subprocess.TimeoutExpired as exc:
194
+ raise EvalError(f"snapshot timed out: {exc}") from exc
195
+ finally:
196
+ # the index and any .lock a timed-out git left beside it (unique per
197
+ # token, so it never blocks a future snapshot — just tidiness)
198
+ index.unlink(missing_ok=True)
199
+ index.with_name(index.name + ".lock").unlink(missing_ok=True)
200
+
201
+
202
+ def _filter_neutral_env(base_git: list[str], env: dict[str, str]) -> dict[str, str]:
203
+ """GIT_CONFIG_* env that overrides every configured filter driver to a
204
+ passthrough and disables attribute files. Robust where `-c` is not:
205
+ keys/values are separate env vars, so a driver name with '=' or dots is
206
+ handled correctly. Used by BOTH the snapshot (index ops) and the job
207
+ script's archive — the repo config and .gitattributes are agent-written
208
+ on both paths."""
209
+ pairs: list[tuple[str, str]] = [("core.attributesFile", "/dev/null")]
210
+ listing = subprocess.run(
211
+ [*base_git, "config", "-z", "--get-regexp", r"^filter\..*\.(clean|smudge|process)$"],
212
+ env=_git_env(env),
213
+ capture_output=True,
214
+ text=True,
215
+ timeout=30,
216
+ check=False,
217
+ )
218
+ # -z: NUL-separated records, each "key\nvalue" — so a value containing a
219
+ # newline can never masquerade as a second record.
220
+ for record in listing.stdout.split("\0"):
221
+ key = record.split("\n", 1)[0]
222
+ if not key.startswith("filter.") or "." not in key[len("filter.") :]:
223
+ continue
224
+ driver = key[len("filter.") : key.rindex(".")]
225
+ pairs += [
226
+ (f"filter.{driver}.clean", "cat"),
227
+ (f"filter.{driver}.smudge", "cat"),
228
+ (f"filter.{driver}.process", ""),
229
+ ]
230
+ out = {"GIT_CONFIG_COUNT": str(len(pairs))}
231
+ for i, (k, v) in enumerate(pairs):
232
+ out[f"GIT_CONFIG_KEY_{i}"] = k
233
+ out[f"GIT_CONFIG_VALUE_{i}"] = v
234
+ return out
235
+
236
+
237
+ def drop_snapshot(ws: Workspace, snapshot: Snapshot) -> None:
238
+ """Release the retaining ref (the commit becomes gc-eligible again). Called
239
+ once the eval result has been read. Best-effort — it never RAISES, so a
240
+ caller's ending sequence cannot hinge on it — but not SILENT: a ref that
241
+ fails to drop keeps its commit alive forever, so the failure is logged.
242
+ A session-reshaped .git is not written to at all (a symlinked refs dir
243
+ would carry the ref deletion outside the workspace): logged and left."""
244
+ try:
245
+ ensure_regular_git_dir(Path(ws.root))
246
+ except GitError as exc:
247
+ log.warning("snapshot ref %s not dropped: %s", snapshot.ref, exc)
248
+ return
249
+ try:
250
+ result = subprocess.run(
251
+ ["git", "-C", str(ws.root), *SAFE_GIT_FLAGS, "update-ref", "-d", snapshot.ref],
252
+ env=_git_env({}),
253
+ capture_output=True,
254
+ text=True,
255
+ timeout=30,
256
+ check=False,
257
+ )
258
+ if result.returncode != 0:
259
+ log.warning("snapshot ref drop failed for %s: %s", snapshot.ref, result.stderr[:200])
260
+ except Exception as exc:
261
+ log.warning("snapshot ref drop errored for %s: %s", snapshot.ref, exc)
262
+
263
+
264
+ def _git_env(extra: dict[str, str]) -> dict[str, str]:
265
+ import os
266
+
267
+ # No HOME, and global/system config pinned to /dev/null: git must honor
268
+ # ONLY the repo's local config while operating on an agent-written tree.
269
+ # A globally-configured filter driver (in the user's ~/.gitconfig or the
270
+ # system config) could otherwise be SELECTED by an agent-authored
271
+ # .gitattributes and execute on the host — dropping HOME alone misses the
272
+ # system config and XDG paths.
273
+ env = {k: os.environ[k] for k in ("PATH", "LANG") if k in os.environ}
274
+ env["GIT_CONFIG_GLOBAL"] = "/dev/null"
275
+ env["GIT_CONFIG_SYSTEM"] = "/dev/null"
276
+ env.update(extra)
277
+ return env
278
+
279
+
280
+ def write_eval_job(
281
+ run_dir: Path,
282
+ name: str,
283
+ *,
284
+ repo_root: Path,
285
+ snapshot_sha: str,
286
+ command: str,
287
+ image: str,
288
+ apptainer_binary: str = "apptainer",
289
+ extra_env: dict[str, str] | None = None,
290
+ artifacts: tuple[str, ...] = (),
291
+ artifact_max_bytes: int = 0,
292
+ gpus: int = 0,
293
+ array: int = 1,
294
+ seed_cache: Path | None = None,
295
+ ) -> Path:
296
+ """Write the orchestrator-authored job script for one dispatched eval.
297
+
298
+ `seed_cache` names the kernel-warmed seed for this target
299
+ (docs/design/eval-cache.md): the job copies its contents into its own
300
+ scratch cache before `uv` runs — a copy, never a bind or a hardlink, so the
301
+ job's cache is its own and the seed is never written; a missing seed or a
302
+ failed copy costs a download, never the eval.
303
+
304
+ `array` > 1 writes ONE script for a Slurm job array: each task derives its
305
+ own job dir `eval-<name>.<k>` from SLURM_ARRAY_TASK_ID (SWEEP_INDEX under a
306
+ local run) and sees the index as SWEEP_INDEX; the task dirs are created
307
+ and cleared here, the script lives under `eval-<name>/`.
308
+
309
+ `gpus` > 0 adds `--nv` to the jail so the job's allocated GPUs (the
310
+ JobSpec requests them) are visible inside the container; nothing else
311
+ about the containment changes.
312
+
313
+ Trust layout, matching the in-job evaluator exactly:
314
+ * the contract COMMAND goes into its own file, read back with
315
+ `sh -c "$(cat ...)"` INSIDE the jail — it never crosses sbatch or
316
+ shell quoting, and it executes only inside apptainer
317
+ --containall/--cleanenv with worktree-only binds;
318
+ * stdout/stderr are captured OUTSIDE the containment into the run
319
+ directory (the jailed process never sees the run dir);
320
+ * env is the evaluator's allowlist shape: uv cache + private venv on
321
+ node-local scratch, plus the call site's extra_env (paired seeds)
322
+ exported as APPTAINERENV_*.
323
+ Returns the script path; the caller submits it via JobSpec(script=...).
324
+
325
+ With `artifacts` (author-syscall launches, research-loop-buildout.md
326
+ Phase A), each declared repo-relative FILE the jailed command produced is
327
+ copied out of the throwaway tree into `<job dir>/artifacts/` — outside the
328
+ jail, after the command, size-capped at `artifact_max_bytes` — with every
329
+ skip recorded in `artifacts.log`. Callers validate the paths (relative, no
330
+ traversal) before passing them; this writer additionally quotes them so
331
+ they cross the script boundary inert.
332
+ """
333
+ ev = run_dir / f"eval-{name}"
334
+ ev.mkdir(parents=True, exist_ok=True)
335
+ task_dirs = [run_dir / f"eval-{name}.{k}" for k in range(array)] if array > 1 else [ev]
336
+ for task_dir in task_dirs:
337
+ task_dir.mkdir(parents=True, exist_ok=True)
338
+ # a resubmitted eval must never be read as its predecessor: every prior
339
+ # artifact — including a leftover extracted tree — goes before submission
340
+ for stale in ("exit-code", "stdout", "stderr", "setup.log", "submitted", "artifacts.log"):
341
+ (task_dir / stale).unlink(missing_ok=True)
342
+ shutil.rmtree(task_dir / "tree", ignore_errors=True)
343
+ shutil.rmtree(task_dir / "artifacts", ignore_errors=True)
344
+ (task_dir / "command.txt").write_text(command)
345
+ # extra_env matches the in-job evaluator's contract: managed keys (HOME,
346
+ # UV_*, PATH...) are DROPPED, never allowed to override the isolation, and
347
+ # keys must be shell-identifier shaped (they are exported unquoted).
348
+ injected = {
349
+ k: v
350
+ for k, v in (extra_env or {}).items()
351
+ if _SHELL_IDENT.match(k) and not managed_eval_env(k)
352
+ }
353
+ safe_git = " ".join(shlex.quote(f) for f in SAFE_GIT_FLAGS)
354
+ # the checkout runs on the agent-written repo too: same filter neutralizers
355
+ # as the snapshot, injected as GIT_CONFIG_* env (robust to '=' in a driver
356
+ # name, unlike -c) so a smudge filter cannot execute during checkout
357
+ neutral = _filter_neutral_env(["git", "-C", str(repo_root), *SAFE_GIT_FLAGS], {})
358
+ if array > 1:
359
+ # the task picks its own job dir; under Slurm the index is the array
360
+ # task id, under a local run the caller exports SWEEP_INDEX
361
+ ev_lines = [
362
+ 'TASK="${SLURM_ARRAY_TASK_ID:-${SWEEP_INDEX:-0}}"',
363
+ f'EV={shlex.quote(str(ev))}."$TASK"',
364
+ 'export SWEEP_INDEX="$TASK" APPTAINERENV_SWEEP_INDEX="$TASK"',
365
+ ]
366
+ else:
367
+ ev_lines = [f"EV={shlex.quote(str(ev))}"]
368
+ lines = [
369
+ "#!/bin/sh",
370
+ "set -u",
371
+ *ev_lines,
372
+ f"REPO={shlex.quote(str(repo_root))}",
373
+ # the extracted tree lives on NODE-LOCAL scratch, not the shared run
374
+ # dir: it dies with the job (nothing to reap on the shared FS), and
375
+ # the wake needs only stdout/exit-code, which stay in $EV
376
+ 'SCRATCH="${SLURM_TMPDIR:-${TMPDIR:-/tmp}}/dispatch-eval-$$"',
377
+ 'TREE="$SCRATCH/tree"',
378
+ 'mkdir -p "$SCRATCH/cache" "$SCRATCH/home" "$SCRATCH/work" "$TREE"',
379
+ # trap FIRST, before anything that can exit, so $SCRATCH never leaks.
380
+ # prune reaps the stale worktree admin entry in $REPO/.git/worktrees
381
+ # left when we deleted $TREE/.git (worktree remove can't run without it)
382
+ 'cleanup() { rm -rf "$SCRATCH"; '
383
+ f'git -C "$REPO" {safe_git} worktree prune >/dev/null 2>&1 || true; }}',
384
+ "trap 'cleanup' EXIT",
385
+ "trap 'echo 143 > \"$EV/exit-code\"; cleanup; trap - EXIT; exit 0' TERM INT HUP",
386
+ # the command file must exist and be non-empty, or sh -c "" would
387
+ # exit 0 with empty output and read as a clean eval that measured
388
+ # nothing
389
+ '[ -s "$EV/command.txt" ] || { echo 96 > "$EV/exit-code"; exit 0; }',
390
+ # git on the node runs with the job user's HOME present, so pin global
391
+ # AND system config to /dev/null: a globally-configured filter driver
392
+ # could otherwise be selected by the agent-authored .gitattributes and
393
+ # execute here, outside the jail. Local repo config still applies and
394
+ # is neutralized by the GIT_CONFIG_KEY_* overrides below.
395
+ "export GIT_CONFIG_GLOBAL=/dev/null GIT_CONFIG_SYSTEM=/dev/null",
396
+ ]
397
+ for k, v in neutral.items():
398
+ lines.append(f"export {k}={shlex.quote(v)}")
399
+ if seed_cache is not None:
400
+ q = shlex.quote(str(seed_cache))
401
+ lines.append(
402
+ f'if [ -d {q} ]; then cp -a {q}/. "$SCRATCH/cache"/ 2>> "$EV/setup.log" || true; fi'
403
+ )
404
+ lines += [
405
+ # Materialize the snapshot by CHECKOUT, not `git archive`: a checkout
406
+ # reproduces content faithfully — INCLUDING .gitattributes — and does
407
+ # NOT apply export-ignore/export-subst (archive-only), so the measured
408
+ # tree equals both the workspace (fidelity) and the fingerprint
409
+ # (integrity). Smudge filters during checkout are neutralized by the
410
+ # GIT_CONFIG_* env above. The worktree's .git gitfile (which would
411
+ # point into $REPO/.git/worktrees, unbound in the jail) is DELETED
412
+ # after checkout, leaving a plain directory; the stale admin entry is
413
+ # pruned on cleanup.
414
+ f'if git -C "$REPO" {safe_git} worktree add --detach "$TREE" '
415
+ f'{shlex.quote(snapshot_sha)} >> "$EV/setup.log" 2>&1; then '
416
+ 'rm -f "$TREE/.git"; ' # plain dir now — nothing points back into $REPO
417
+ 'else echo 97 > "$EV/exit-code"; exit 0; fi',
418
+ 'export UV_CACHE_DIR="$SCRATCH/cache" UV_LINK_MODE=copy '
419
+ 'UV_PROJECT_ENVIRONMENT="$SCRATCH/cache/venv"',
420
+ 'export APPTAINERENV_UV_CACHE_DIR="$UV_CACHE_DIR" '
421
+ "APPTAINERENV_UV_LINK_MODE=copy "
422
+ 'APPTAINERENV_UV_PROJECT_ENVIRONMENT="$UV_PROJECT_ENVIRONMENT"',
423
+ ]
424
+ for key, value in injected.items():
425
+ lines.append(f"export {key}={shlex.quote(value)} APPTAINERENV_{key}={shlex.quote(value)}")
426
+ if image:
427
+ lines += [
428
+ # the jail: identical flags to SubprocessEvaluator._run; stdout is
429
+ # redirected OUTSIDE apptainer, so the result lands in the run dir
430
+ # without the jailed process ever seeing it
431
+ f"{shlex.quote(apptainer_binary)} exec --containall --cleanenv "
432
+ + ("--nv " if gpus > 0 else "")
433
+ + '--bind "$TREE:$TREE" --home "$SCRATCH/home:$SCRATCH/home" '
434
+ '--bind "$SCRATCH/cache:$SCRATCH/cache" --pwd "$TREE" '
435
+ # /tmp and /var/tmp inside the jail live on node-local scratch,
436
+ # not apptainer's small tmpfs: a command that writes compile caches
437
+ # or checkpoints to /tmp (any training script that was not told
438
+ # otherwise) must not die on a temp-disk limit
439
+ '--workdir "$SCRATCH/work" '
440
+ f"{shlex.quote(image)} "
441
+ 'sh -c "$(cat "$EV/command.txt")" '
442
+ '> "$EV/stdout" 2> "$EV/stderr"',
443
+ 'echo $? > "$EV/exit-code"',
444
+ ]
445
+ else:
446
+ # UNCONTAINED (dev/tests, no image): run directly in the throwaway
447
+ # tree under `env -i` with the same allowlist shape the uncontained
448
+ # evaluator used — the submitting process's env (which can hold live
449
+ # keys) must never reach the agent-authored command
450
+ bare_env = " ".join(
451
+ [
452
+ 'HOME="$SCRATCH/home"',
453
+ 'PATH="$PATH"',
454
+ 'LANG="${LANG:-C.UTF-8}"',
455
+ 'TMPDIR="$SCRATCH"',
456
+ 'UV_CACHE_DIR="$UV_CACHE_DIR"',
457
+ "UV_LINK_MODE=copy",
458
+ 'UV_PROJECT_ENVIRONMENT="$UV_PROJECT_ENVIRONMENT"',
459
+ ]
460
+ + [f"{key}={shlex.quote(value)}" for key, value in injected.items()]
461
+ )
462
+ lines += [
463
+ f'cd "$TREE" && env -i {bare_env} '
464
+ 'sh -c "$(cat "$EV/command.txt")" '
465
+ '> "$EV/stdout" 2> "$EV/stderr"',
466
+ 'echo $? > "$EV/exit-code"',
467
+ ]
468
+ if artifacts:
469
+ # copy-out runs OUTSIDE the jail, after the command: only declared,
470
+ # caller-validated repo-relative FILES, each size-capped; every skip is
471
+ # recorded so the wake can tell the author exactly what happened.
472
+ # Two hostile shapes to defeat, since the jailed (agent) command wrote
473
+ # this tree and picked these names:
474
+ # * INJECTION — a name like `$(cmd)` must never reach the shell as
475
+ # code. Every path goes in single-quoted (shlex.quote) argument
476
+ # position, and every log line is a single-quoted LITERAL (a name
477
+ # interpolated into a double-quoted `echo` would execute), so the
478
+ # name is inert everywhere.
479
+ # * ESCAPE via symlink — a declared file (or an intermediate dir) may
480
+ # be a symlink to a host file; the host-side cp would dereference
481
+ # it. `realpath` resolves the whole path and we copy ONLY when the
482
+ # resolved target stays under $TREE (and copy the RESOLVED path, so
483
+ # there is no resolve-then-copy gap). The command has already
484
+ # finished, so the tree is quiescent — no TOCTOU.
485
+ lines.append('mkdir -p "$EV/artifacts"')
486
+ lines.append('TREE_REAL=$(realpath "$TREE" 2>/dev/null || echo "$TREE")')
487
+ for art in artifacts:
488
+ q = shlex.quote(art) # safe in argument position (single-quoted)
489
+ skip_type = shlex.quote(f"skipped (not a regular file in the tree): {art}")
490
+ skip_big = shlex.quote(f"skipped (over {int(artifact_max_bytes)} bytes): {art}")
491
+ fail_cp = shlex.quote(f"copy failed: {art}")
492
+ lines.append(
493
+ f'AP=$(realpath "$TREE"/{q} 2>/dev/null || true); '
494
+ f'case "$AP" in "$TREE_REAL"/*) '
495
+ f'if [ -f "$AP" ] && [ "$(wc -c < "$AP")" -le {int(artifact_max_bytes)} ]; then '
496
+ f'mkdir -p "$EV/artifacts/$(dirname {q})" && cp "$AP" "$EV/artifacts"/{q} '
497
+ f'|| echo {fail_cp} >> "$EV/artifacts.log"; '
498
+ f'elif [ -f "$AP" ]; then echo {skip_big} >> "$EV/artifacts.log"; '
499
+ f'else echo {skip_type} >> "$EV/artifacts.log"; fi ;; '
500
+ f'*) echo {skip_type} >> "$EV/artifacts.log" ;; esac'
501
+ )
502
+ lines += [
503
+ "exit 0",
504
+ ]
505
+ script = ev / "job.sh"
506
+ script.write_text("\n".join(lines) + "\n")
507
+ script.chmod(0o755)
508
+ return script
509
+
510
+
511
+ def eval_job_spec(
512
+ script: Path,
513
+ *,
514
+ job_name: str,
515
+ account: str,
516
+ partition: str,
517
+ eval_minutes: int,
518
+ cpus: int = 4,
519
+ mem: str = "8G",
520
+ gpus: int = 0,
521
+ nice: int = 0,
522
+ array: str = "",
523
+ ) -> JobSpec:
524
+ """The JobSpec for one dispatched eval: the hint CLAMPED to our ceiling
525
+ plus setup slack — a contract value above EVAL_JOB_MINUTES_CEILING must
526
+ not create a longer Slurm job than the ceiling allows. `gpus` is the
527
+ benchmark's contract field; the caller has already placed the job on
528
+ the GPU lane (DispatchSettings.placement) when it is nonzero, and the
529
+ job is sized for it: a GPU eval gets at least EVAL_CPUS_PER_GPU cores
530
+ and EVAL_MEM_GB_PER_GPU GB per GPU (a training eval's data loading and
531
+ torch.compile workers do not fit the CPU eval's 4 cores / 8 GB). `nice`
532
+ lowers the job's priority below the kernel's evals (the launcher sets it)."""
533
+ if gpus > 0:
534
+ cpus = max(cpus, EVAL_CPUS_PER_GPU * gpus)
535
+ given = _mem_gb(mem)
536
+ # an explicit request is never SHRUNK: only a parseable value below
537
+ # the per-GPU floor is raised; anything unparseable passes through
538
+ if given is not None and given < EVAL_MEM_GB_PER_GPU * gpus:
539
+ mem = f"{EVAL_MEM_GB_PER_GPU * gpus}G"
540
+ return JobSpec(
541
+ job_name=job_name[:60],
542
+ account=account,
543
+ partition=partition,
544
+ time_minutes=effective_eval_minutes(eval_minutes) + EVAL_JOB_SETUP_MINUTES,
545
+ script=str(script),
546
+ cpus=cpus,
547
+ mem=mem,
548
+ gpus=gpus,
549
+ nice=nice,
550
+ array=array,
551
+ )
552
+
553
+
554
+ def _mem_gb(mem: str) -> int | None:
555
+ """A Slurm --mem value ("8G", "512M", "1T", "16") in whole GB, rounded
556
+ down; None when unparseable (the caller then leaves it alone)."""
557
+ text = mem.strip().upper()
558
+ scale = {"K": 1 / (1024 * 1024), "M": 1 / 1024, "G": 1.0, "T": 1024.0}
559
+ unit = text[-1:] if text[-1:] in scale else ""
560
+ number = text[:-1] if unit else text
561
+ try:
562
+ value = float(number)
563
+ except ValueError:
564
+ return None
565
+ if not math.isfinite(value) or value < 0:
566
+ return None # "nanG"/"infG": not a size — pass it through untouched
567
+ return int(value * (scale[unit] if unit else 1 / 1024)) # bare Slurm --mem is MB
568
+
569
+
570
+ def read_eval_result(run_dir: Path, name: str, metric: str) -> float:
571
+ """The wake side of one dispatched eval. Raises EvalError with the same
572
+ semantics as the in-job evaluator: nonzero exit or an unreadable metric
573
+ is an eval failure (outcome `eval-error`, ending `aborted`), and a
574
+ MISSING exit-code file means the job died before the wrapper ran —
575
+ also a failure, never a silent skip."""
576
+ ev = run_dir / f"eval-{name}"
577
+ try:
578
+ code = int((ev / "exit-code").read_text().strip())
579
+ except (OSError, ValueError) as exc:
580
+ raise EvalError(f"dispatched eval {name}: no exit code ({exc})") from exc
581
+ stdout = ""
582
+ with contextlib.suppress(OSError, ValueError):
583
+ stdout = (ev / "stdout").read_text(errors="replace")
584
+ if code != 0:
585
+ tail = ""
586
+ with contextlib.suppress(OSError, ValueError):
587
+ tail = (ev / "stderr").read_text(errors="replace")[-300:]
588
+ raise EvalError(f"dispatched eval {name} failed ({code}): {tail}")
589
+ value = metric_from_output(stdout, metric)
590
+ if value is None:
591
+ raise EvalError(f"dispatched eval {name}: no readable {metric!r} in output")
592
+ if not math.isfinite(value):
593
+ # same rule as the in-job evaluator: json parses bare NaN/Infinity,
594
+ # and a NaN score entering a comparison is worse than a failure
595
+ raise EvalError(f"dispatched eval {name}: non-finite {metric!r} ({value})")
596
+ return value
597
+
598
+
599
+ def result_summary(run_dir: Path, name: str) -> str:
600
+ """One line for reports/logs; never raises."""
601
+ ev = run_dir / f"eval-{name}"
602
+ try:
603
+ code = (ev / "exit-code").read_text(errors="replace").strip()
604
+ except (OSError, ValueError):
605
+ code = "?"
606
+ try:
607
+ last = [
608
+ ln for ln in (ev / "stdout").read_text(errors="replace").splitlines() if ln.strip()
609
+ ][-1]
610
+ except (OSError, ValueError, IndexError):
611
+ last = ""
612
+ return f"eval-{name}: exit={code} {last[:160]}"
613
+
614
+
615
+ def parse_result_json(run_dir: Path, name: str) -> dict:
616
+ """The full JSON line (margins, per-encoder blocks) for report embedding;
617
+ empty dict when absent."""
618
+ ev = run_dir / f"eval-{name}"
619
+ try:
620
+ for line in reversed((ev / "stdout").read_text(errors="replace").splitlines()):
621
+ line = line.strip()
622
+ if line.startswith("{"):
623
+ try:
624
+ data = json.loads(line)
625
+ except json.JSONDecodeError:
626
+ continue
627
+ if isinstance(data, dict):
628
+ return data
629
+ except OSError:
630
+ pass
631
+ return {}