outerloop-science 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. outerloop/__init__.py +18 -0
  2. outerloop/__main__.py +3 -0
  3. outerloop/appauth.py +230 -0
  4. outerloop/appmanifest.py +203 -0
  5. outerloop/attempt.py +3784 -0
  6. outerloop/brief.py +528 -0
  7. outerloop/cli.py +621 -0
  8. outerloop/climbboard.py +1395 -0
  9. outerloop/compute.py +654 -0
  10. outerloop/contract.py +492 -0
  11. outerloop/contract_cli.py +63 -0
  12. outerloop/disk.py +164 -0
  13. outerloop/dispatch.py +631 -0
  14. outerloop/evalcache.py +147 -0
  15. outerloop/followup.py +2172 -0
  16. outerloop/github.py +1531 -0
  17. outerloop/harness.py +1435 -0
  18. outerloop/housekeeping.py +151 -0
  19. outerloop/image.py +368 -0
  20. outerloop/init.py +744 -0
  21. outerloop/intake.py +126 -0
  22. outerloop/launchlog.py +239 -0
  23. outerloop/limits.py +80 -0
  24. outerloop/maintain.py +353 -0
  25. outerloop/maintain_agent_cli.py +81 -0
  26. outerloop/maintain_post_cli.py +140 -0
  27. outerloop/markers.py +48 -0
  28. outerloop/measure.py +529 -0
  29. outerloop/orchestrator.py +2011 -0
  30. outerloop/panel.py +188 -0
  31. outerloop/paths.py +40 -0
  32. outerloop/posting.py +160 -0
  33. outerloop/progress.py +170 -0
  34. outerloop/py.typed +0 -0
  35. outerloop/review.py +615 -0
  36. outerloop/review_agent.py +263 -0
  37. outerloop/review_agent_cli.py +209 -0
  38. outerloop/review_post_cli.py +162 -0
  39. outerloop/review_summarize_cli.py +165 -0
  40. outerloop/role_runner.py +229 -0
  41. outerloop/roles.py +274 -0
  42. outerloop/rolespec.py +91 -0
  43. outerloop/runstate.py +385 -0
  44. outerloop/steward.py +845 -0
  45. outerloop/style.py +12 -0
  46. outerloop/syscall.py +1192 -0
  47. outerloop/syscall_cli.py +762 -0
  48. outerloop/tick.py +3422 -0
  49. outerloop/verifier.py +403 -0
  50. outerloop/verify_agent.py +151 -0
  51. outerloop/verify_agent_cli.py +95 -0
  52. outerloop/verify_post_cli.py +116 -0
  53. outerloop/watcher.py +203 -0
  54. outerloop_science-0.1.0.dist-info/METADATA +152 -0
  55. outerloop_science-0.1.0.dist-info/RECORD +59 -0
  56. outerloop_science-0.1.0.dist-info/WHEEL +4 -0
  57. outerloop_science-0.1.0.dist-info/entry_points.txt +2 -0
  58. outerloop_science-0.1.0.dist-info/licenses/LICENSE +202 -0
  59. outerloop_science-0.1.0.dist-info/licenses/NOTICE +5 -0
outerloop/compute.py ADDED
@@ -0,0 +1,654 @@
1
+ """Compute behind one small interface: submit, status, cancel.
2
+
3
+ Everything the loop knows about compute goes through these verbs, so a
4
+ backend is one implementation: `SlurmCompute` submits real cluster jobs;
5
+ `LocalCompute` runs the same job specs as subprocesses in the current
6
+ allocation. A CI runner or a cloud/GPU-rental backend would be another
7
+ implementation of the same verbs — the callers never change.
8
+
9
+ The status query preserves a distinction the fail-safe design depends on
10
+ (docs/design/architecture.md, "Wake delivery and fail-safety"): a FAILED
11
+ query ("Slurm unknown") is not the same as a successful query that finds
12
+ nothing ("job gone") — misreading an outage as a vanished job would
13
+ terminate healthy runs.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import contextlib
19
+ import logging
20
+ import os
21
+ import re
22
+ import shlex
23
+ import signal
24
+ import subprocess
25
+ import time
26
+ from collections.abc import Callable, Sequence
27
+ from dataclasses import dataclass, field
28
+ from pathlib import Path
29
+ from typing import Protocol
30
+
31
+ log = logging.getLogger(__name__)
32
+
33
+ # Terminal Slurm states (prefix-matched: sacct reports e.g. "CANCELLED by 123").
34
+ TERMINAL_STATES = (
35
+ "COMPLETED",
36
+ "FAILED",
37
+ "CANCELLED",
38
+ "TIMEOUT",
39
+ "OUT_OF_MEMORY",
40
+ "NODE_FAIL",
41
+ "PREEMPTED",
42
+ "BOOT_FAIL",
43
+ "DEADLINE",
44
+ )
45
+ # A successful query that returns no record: the job left Slurm's memory.
46
+ GONE = "GONE"
47
+
48
+
49
+ class SlurmError(RuntimeError):
50
+ """A Slurm command failed (submit/cancel), with its stderr."""
51
+
52
+
53
+ class SlurmQueryError(RuntimeError):
54
+ """A status query failed — the answer is UNKNOWN, not 'job gone'.
55
+
56
+ Callers must treat this as "defer and retry", never as a terminal state.
57
+ """
58
+
59
+
60
+ @dataclass(frozen=True)
61
+ class CommandResult:
62
+ returncode: int
63
+ stdout: str
64
+ stderr: str
65
+
66
+
67
+ Runner = Callable[[Sequence[str], int], CommandResult]
68
+
69
+
70
+ def _subprocess_runner(argv: Sequence[str], timeout_s: int) -> CommandResult:
71
+ completed = subprocess.run(
72
+ list(argv), capture_output=True, text=True, timeout=timeout_s, check=False
73
+ )
74
+ return CommandResult(completed.returncode, completed.stdout, completed.stderr)
75
+
76
+
77
+ @dataclass(frozen=True)
78
+ class JobSpec:
79
+ """One sbatch submission. `command` is run via --wrap; a script path can
80
+ be passed as `script` instead (mutually exclusive).
81
+
82
+ --wrap executes under a shell on the compute node: `command` must be
83
+ built from trusted parts, with anything variable passed through
84
+ `quote_command`. Never interpolate agent- or contract-supplied text."""
85
+
86
+ job_name: str
87
+ account: str
88
+ partition: str
89
+ time_minutes: int
90
+ command: str = ""
91
+ script: str = ""
92
+ script_args: tuple[str, ...] = ()
93
+ cpus: int = 1
94
+ mem: str = "2G"
95
+ gpus: int = 0
96
+ qos: str = ""
97
+ # a positive nice LOWERS priority (Slurm, like Unix): experiments yield to
98
+ # the kernel's own evals and re-measures when a slot frees
99
+ nice: int = 0
100
+ output: str = "/dev/null"
101
+ # Slurm scheduling controls
102
+ dependency: str = "" # e.g. "afterany:12345" or "singleton"
103
+ begin: str = "" # e.g. "now+30" or an absolute "YYYY-MM-DDTHH:MM:SS"
104
+ # a job array: "0-15%4" runs tasks 0..15, at most 4 at a time; the queue
105
+ # holds one entry and squeue names its tasks `<id>_<k>`
106
+ array: str = ""
107
+ extra: tuple[str, ...] = ()
108
+
109
+ def to_argv(self) -> list[str]:
110
+ if bool(self.command) == bool(self.script):
111
+ raise ValueError("exactly one of command/script must be set")
112
+ argv = [
113
+ "sbatch",
114
+ "--parsable",
115
+ f"--job-name={self.job_name}",
116
+ f"--time={self.time_minutes}",
117
+ f"--cpus-per-task={self.cpus}",
118
+ f"--mem={self.mem}",
119
+ f"--output={self.output}",
120
+ ]
121
+ if self.account: # unset bills the caller's default Slurm association
122
+ argv.append(f"--account={self.account}")
123
+ if self.partition: # unset lets Slurm pick its default partition
124
+ argv.append(f"--partition={self.partition}")
125
+ if self.gpus:
126
+ # per-NODE, not per-job (--gpus): every job here is single-node,
127
+ # and Slurm submit plugins commonly classify a job by its
128
+ # per-node GRES — the per-job form has been rejected on a GPU
129
+ # partition as "CPU job setup is not valid"
130
+ argv.append(f"--gpus-per-node={self.gpus}")
131
+ if self.qos:
132
+ argv.append(f"--qos={self.qos}")
133
+ if self.nice:
134
+ argv.append(f"--nice={self.nice}")
135
+ if self.dependency:
136
+ argv.append(f"--dependency={self.dependency}")
137
+ if self.begin:
138
+ argv.append(f"--begin={self.begin}")
139
+ if self.array:
140
+ argv.append(f"--array={self.array}")
141
+ argv.extend(self.extra)
142
+ if self.command:
143
+ argv.append(f"--wrap={self.command}")
144
+ else:
145
+ argv.append(self.script)
146
+ argv.extend(self.script_args)
147
+ return argv
148
+
149
+
150
+ # `reason`, `gres` and `limit` feed the queue view: why a job waits, what it holds
151
+ # waits, whether it holds GPUs); the board reads the first six by key
152
+ QUEUE_FIELDS = (
153
+ "id",
154
+ "name",
155
+ "state",
156
+ "elapsed",
157
+ "partition",
158
+ "submitted",
159
+ "reason",
160
+ "gres",
161
+ "limit",
162
+ )
163
+
164
+
165
+ class Compute(Protocol):
166
+ """The verbs every compute backend implements. Callers (the measurer, the
167
+ launcher, the wake dispatcher) depend on this, never on a backend."""
168
+
169
+ def submit(self, spec: JobSpec) -> str: ...
170
+ def status(self, job_id: str) -> str: ...
171
+ def pending_reason(self, job_id: str) -> str: ...
172
+ def job_partition(self, job_id: str) -> str: ...
173
+ def active_job_names(self) -> list[str]: ...
174
+ def queue_snapshot(self) -> list[dict[str, str]]: ...
175
+ def lane_load(self, partition: str) -> dict[str, int]: ...
176
+ def job_id_for_name(self, name: str) -> str: ...
177
+ def cancel(self, job_id: str) -> bool: ...
178
+
179
+
180
+ def local_mode() -> bool:
181
+ """OUTERLOOP_COMPUTE=local selects the monolith: every job a
182
+ synchronous subprocess of the caller (docs/design/onboarding.md) — the
183
+ zero-cluster on-ramp and the paper's serialized-baseline ablation. Any
184
+ other value (or none) is Slurm. This helper is the only reader of the
185
+ env var, so mode checks cannot drift."""
186
+ return os.environ.get("OUTERLOOP_COMPUTE", "").strip().lower() == "local"
187
+
188
+
189
+ def compute_from_env() -> SlurmCompute | LocalCompute:
190
+ """The deployment's compute backend, per `local_mode`."""
191
+ return LocalCompute() if local_mode() else SlurmCompute()
192
+
193
+
194
+ _JOB_ID = re.compile(r"^\d+(_\d+)?$") # a job, or one task of a job array (`<id>_<k>`)
195
+
196
+
197
+ def _check_job_id(job_id: str) -> None:
198
+ if not _JOB_ID.match(job_id):
199
+ raise ValueError(f"not a job id: {job_id!r}")
200
+
201
+
202
+ def array_indices(spec: str) -> list[int]:
203
+ """The task indices of an array spec: "0-15%4" -> 0..15 (the %K throttle
204
+ is the scheduler's concern); "" -> none."""
205
+ body = spec.split("%", 1)[0].strip()
206
+ if not body:
207
+ return []
208
+ lo, sep, hi = body.partition("-")
209
+ if not sep:
210
+ return [int(lo)] if lo.isdigit() else []
211
+ if not (lo.isdigit() and hi.isdigit()):
212
+ return []
213
+ return list(range(int(lo), int(hi) + 1))
214
+
215
+
216
+ def combine_states(states: Sequence[str]) -> str:
217
+ """One state for a job array from its tasks' states: running while any
218
+ task runs, pending while any task waits, terminal only when every task
219
+ is — COMPLETED if all are, else the first other terminal state (FAILED,
220
+ TIMEOUT, CANCELLED...), so a sweep with one dead task reads as failed."""
221
+ if len(states) == 1:
222
+ return states[0]
223
+ for want in ("RUNNING", "COMPLETING"):
224
+ if any(s.startswith(want) for s in states):
225
+ return want
226
+ if any(is_pending(s) for s in states):
227
+ return "PENDING"
228
+ live = [s for s in states if not is_terminal(s)]
229
+ if live:
230
+ return live[0]
231
+ bad = [s for s in states if not s.startswith("COMPLETED")]
232
+ return bad[0] if bad else "COMPLETED"
233
+
234
+
235
+ @dataclass
236
+ class SlurmCompute:
237
+ """The three verbs, plus afterany for wake jobs."""
238
+
239
+ runner: Runner = field(default=_subprocess_runner)
240
+ command_timeout_s: int = 60
241
+
242
+ def submit(self, spec: JobSpec) -> str:
243
+ """Submit; returns the job id. Raises SlurmError on failure."""
244
+ result = self.runner(spec.to_argv(), self.command_timeout_s)
245
+ if result.returncode != 0:
246
+ raise SlurmError(f"sbatch failed ({result.returncode}): {result.stderr.strip()}")
247
+ job_id = result.stdout.strip().split(";")[0]
248
+ if not job_id.isdigit():
249
+ raise SlurmError(f"sbatch returned no job id: {result.stdout.strip()!r}")
250
+ log.info("submitted %s as job %s", spec.job_name, job_id)
251
+ return job_id
252
+
253
+ def status(self, job_id: str) -> str:
254
+ """The job's Slurm state, or GONE when a *successful* query finds no
255
+ record. Raises SlurmQueryError when the query itself fails."""
256
+ _check_job_id(job_id)
257
+ try:
258
+ result = self.runner(
259
+ ["sacct", "-j", job_id, "--parsable2", "--noheader", "-X", "-o", "State"],
260
+ self.command_timeout_s,
261
+ )
262
+ except (OSError, subprocess.TimeoutExpired) as exc:
263
+ raise SlurmQueryError(f"sacct did not run: {exc}") from exc
264
+ if result.returncode != 0:
265
+ raise SlurmQueryError(f"sacct failed ({result.returncode}): {result.stderr.strip()}")
266
+ # a job array answers one line per task; the array's state is theirs combined
267
+ states = [ln.strip() for ln in result.stdout.splitlines() if ln.strip()]
268
+ return combine_states(states) if states else GONE
269
+
270
+ def elapsed_seconds(self, job_id: str) -> int | None:
271
+ """How long the job actually ran (sacct Elapsed), or None when sacct
272
+ has no record. Raises SlurmQueryError when the query itself fails."""
273
+ _check_job_id(job_id)
274
+ try:
275
+ result = self.runner(
276
+ ["sacct", "-j", job_id, "--parsable2", "--noheader", "-X", "-o", "Elapsed"],
277
+ self.command_timeout_s,
278
+ )
279
+ except (OSError, subprocess.TimeoutExpired) as exc:
280
+ raise SlurmQueryError(f"sacct did not run: {exc}") from exc
281
+ if result.returncode != 0:
282
+ raise SlurmQueryError(f"sacct failed ({result.returncode}): {result.stderr.strip()}")
283
+ text = result.stdout.strip().splitlines()[0].strip() if result.stdout.strip() else ""
284
+ return parse_elapsed(text) if text else None
285
+
286
+ def pending_reason(self, job_id: str) -> str:
287
+ """Why a PENDING job is pending — Slurm's reason (`Dependency`,
288
+ `DependencyNeverSatisfied`, `Priority`, ...), or "" when squeue no
289
+ longer lists it. Raises SlurmQueryError when the query itself fails."""
290
+ _check_job_id(job_id)
291
+ try:
292
+ result = self.runner(["squeue", "-j", job_id, "-h", "-o", "%r"], self.command_timeout_s)
293
+ except (OSError, subprocess.TimeoutExpired) as exc:
294
+ raise SlurmQueryError(f"squeue did not run: {exc}") from exc
295
+ if result.returncode != 0:
296
+ raise SlurmQueryError(f"squeue failed ({result.returncode}): {result.stderr.strip()}")
297
+ return result.stdout.strip().splitlines()[0].strip() if result.stdout.strip() else ""
298
+
299
+ def job_partition(self, job_id: str) -> str:
300
+ """The partition(s) a queued job currently sits in, as squeue prints
301
+ them, or "" when squeue no longer lists it. A site can MOVE a pending
302
+ job off the partition it was submitted to (Torch does, under
303
+ congestion); callers compare this with what they asked for."""
304
+ _check_job_id(job_id)
305
+ try:
306
+ result = self.runner(["squeue", "-j", job_id, "-h", "-o", "%P"], self.command_timeout_s)
307
+ except (OSError, subprocess.TimeoutExpired) as exc:
308
+ raise SlurmQueryError(f"squeue did not run: {exc}") from exc
309
+ if result.returncode != 0:
310
+ raise SlurmQueryError(f"squeue failed ({result.returncode}): {result.stderr.strip()}")
311
+ return result.stdout.strip().splitlines()[0].strip() if result.stdout.strip() else ""
312
+
313
+ def active_job_names(self) -> list[str]:
314
+ """The names of this user's PENDING and RUNNING jobs. Names, not
315
+ commands: squeue's Command field is not guaranteed to carry --wrap
316
+ strings, while %j is always the submitted name. Raises
317
+ SlurmQueryError on failure — callers that delete things keyed on
318
+ this must treat blindness as "delete nothing"."""
319
+ try:
320
+ result = self.runner(
321
+ ["squeue", "--me", "--noheader", "-o", "%j"], self.command_timeout_s
322
+ )
323
+ except (OSError, subprocess.TimeoutExpired) as exc:
324
+ raise SlurmQueryError(f"squeue did not run: {exc}") from exc
325
+ if result.returncode != 0:
326
+ raise SlurmQueryError(f"squeue failed ({result.returncode}): {result.stderr.strip()}")
327
+ return [line.strip() for line in result.stdout.splitlines() if line.strip()]
328
+
329
+ def queue_snapshot(self) -> list[dict[str, str]]:
330
+ """This user's PENDING and RUNNING jobs as rows of QUEUE_FIELDS — what
331
+ a queue view needs, nothing a caller acts on. Raises SlurmQueryError
332
+ on failure, like active_job_names."""
333
+ try:
334
+ result = self.runner(
335
+ ["squeue", "--me", "--noheader", "-o", "%i|%j|%T|%M|%P|%V|%r|%b|%l"],
336
+ self.command_timeout_s,
337
+ )
338
+ except (OSError, subprocess.TimeoutExpired) as exc:
339
+ raise SlurmQueryError(f"squeue did not run: {exc}") from exc
340
+ if result.returncode != 0:
341
+ raise SlurmQueryError(f"squeue failed ({result.returncode}): {result.stderr.strip()}")
342
+ rows: list[dict[str, str]] = []
343
+ for line in result.stdout.splitlines():
344
+ parts = line.strip().split("|")
345
+ if len(parts) == len(QUEUE_FIELDS) and parts[0]:
346
+ rows.append(dict(zip(QUEUE_FIELDS, parts, strict=True)))
347
+ return rows
348
+
349
+ def lane_load(self, partition: str) -> dict[str, int]:
350
+ """Node counts by state on a lane (a partition or a comma-separated
351
+ list), from sinfo: {"idle": 3, "mixed": 20, "allocated": 11}. Context
352
+ for the queue view, nothing a caller acts on. Empty when no lane is
353
+ named; raises SlurmQueryError on a failed query."""
354
+ if not partition:
355
+ return {}
356
+ try:
357
+ result = self.runner(
358
+ ["sinfo", "--noheader", "-p", partition, "-o", "%T %D"], self.command_timeout_s
359
+ )
360
+ except (OSError, subprocess.TimeoutExpired) as exc:
361
+ raise SlurmQueryError(f"sinfo did not run: {exc}") from exc
362
+ if result.returncode != 0:
363
+ raise SlurmQueryError(f"sinfo failed ({result.returncode}): {result.stderr.strip()}")
364
+ load: dict[str, int] = {}
365
+ for line in result.stdout.splitlines():
366
+ parts = line.split()
367
+ if len(parts) != 2 or not parts[1].isdigit():
368
+ continue
369
+ state = parts[0].rstrip("*~#!%$@^-") # sinfo's state flags (draining, no-respond...)
370
+ load[state] = load.get(state, 0) + int(parts[1])
371
+ return load
372
+
373
+ def job_id_for_name(self, name: str) -> str:
374
+ """The id of this user's PENDING/RUNNING job with exactly `name`, or
375
+ "" if none. Authoritative for "is this still live" independent of any
376
+ local bookkeeping — a dispatched job is visible here even if the
377
+ submitter died before recording its id. Raises SlurmQueryError on a
378
+ failed query (the caller must not treat blindness as 'not running')."""
379
+ try:
380
+ result = self.runner(
381
+ ["squeue", "--me", "--name", name, "--noheader", "-o", "%i"],
382
+ self.command_timeout_s,
383
+ )
384
+ except (OSError, subprocess.TimeoutExpired) as exc:
385
+ raise SlurmQueryError(f"squeue did not run: {exc}") from exc
386
+ if result.returncode != 0:
387
+ raise SlurmQueryError(f"squeue failed ({result.returncode}): {result.stderr.strip()}")
388
+ ids = [line.strip() for line in result.stdout.splitlines() if line.strip()]
389
+ return ids[0] if ids else ""
390
+
391
+ def cancel(self, job_id: str) -> bool:
392
+ """Cancel; idempotent (cancelling a finished job is not an error).
393
+ False when scancel itself failed, so a caller that must know (the
394
+ sweep's cancel-on-end) can try again; most callers are best-effort."""
395
+ _check_job_id(job_id)
396
+ result = self.runner(["scancel", job_id], self.command_timeout_s)
397
+ if result.returncode != 0:
398
+ log.warning("scancel %s: %s", job_id, result.stderr.strip())
399
+ return False
400
+ return True
401
+
402
+
403
+ # Local job ids start far above any real Slurm id so the two can never be
404
+ # confused in a record; they stay numeric because callers validate isdigit.
405
+ _LOCAL_JOB_BASE = 9_000_000_000
406
+
407
+
408
+ def _local_state_dir() -> Path | None:
409
+ """Where local job states persist across processes (the tick and the
410
+ attempts it spawns each hold their own LocalCompute): under the state
411
+ root when the deployment names one, else nowhere (memory-only — tests).
412
+ Local jobs are synchronous, so only TERMINAL states ever need sharing."""
413
+ root = os.environ.get("OUTERLOOP_ROOT", "").strip()
414
+ return Path(root) / "local_jobs" if root else None
415
+
416
+
417
+ @dataclass
418
+ class LocalCompute:
419
+ """The same verbs, run as subprocesses in THIS allocation — synchronously:
420
+ `submit` returns with the job already terminal, so a caller that checks
421
+ for the result after submitting finds it on disk and nothing ever parks.
422
+ This is the degenerate backend for evals cheap enough to ride the current
423
+ allocation, for deployments with no cluster at all, and for tests. It runs
424
+ the identical job scripts the cluster runs (fresh checkout of the sealed
425
+ sha, results to the job dir); only WHERE they run differs."""
426
+
427
+ _states: dict[str, str] = field(default_factory=dict)
428
+ _seq: int = 0
429
+ minute_s: int = 60 # a walltime minute; tests shrink it to exercise the kill
430
+
431
+ def submit(self, spec: JobSpec) -> str:
432
+ if bool(spec.command) == bool(spec.script):
433
+ # same contract SlurmCompute enforces via to_argv
434
+ raise ValueError("exactly one of command/script must be set")
435
+ argv = ["sh", spec.script, *spec.script_args] if spec.script else ["sh", "-c", spec.command]
436
+ self._seq += 1
437
+ # unique across processes: the tick and its attempts each count from 1.
438
+ # A million-wide slot per (pid mod 10k); exhausting it fails LOUD —
439
+ # a silent wraparound would let one process read another's terminal
440
+ # state under a reused id.
441
+ if self._seq >= 1_000_000:
442
+ raise SlurmError("local job id space exhausted for this process")
443
+ job_id = str(_LOCAL_JOB_BASE + (os.getpid() % 10_000) * 1_000_000 + self._seq)
444
+
445
+ # An explicit env allowlist:
446
+ # the submitting process holds live keys (and any inherited
447
+ # APPTAINERENV_* would cross --cleanenv into the container), so the
448
+ # job script starts from a minimal environment and sets its own.
449
+ # OUTERLOOP_* / REVIEW_HERMES_* pass through as a PREFIX rule:
450
+ # Slurm jobs inherit the tick's whole environment, and local jobs
451
+ # need the same config surface (compute mode, author backend, panel,
452
+ # key-file PATHS). Enumerating allowed names is how a mode flag dies
453
+ # silently (terra #222/#223) — but VALUE-bearing secret names under
454
+ # the prefix (a *_PAT / *_TOKEN / *_KEY, as opposed to a *_KEY_FILE
455
+ # path) must never reach a job that runs untrusted evaluation code.
456
+ def _secret_name(name: str) -> bool:
457
+ return name.endswith(("_PAT", "_TOKEN", "_SECRET", "_PASSWORD", "_KEY"))
458
+
459
+ job_env = {
460
+ k: v
461
+ for k, v in os.environ.items()
462
+ if k in ("PATH", "HOME", "LANG", "TMPDIR", "SLURM_TMPDIR", "USER", "LOGNAME")
463
+ or (k.startswith(("OUTERLOOP_", "REVIEW_HERMES_")) and not _secret_name(k))
464
+ }
465
+ indices = array_indices(spec.array)
466
+ if indices:
467
+ # a job array runs its tasks in turn — there is no queue here to
468
+ # throttle; each task keeps its own state and output under
469
+ # `<id>_<k>`, and the array's own state is theirs combined
470
+ states = [
471
+ self._run_and_record(
472
+ spec, argv, {**job_env, "SLURM_ARRAY_TASK_ID": str(i)}, f"{job_id}_{i}"
473
+ )
474
+ for i in indices
475
+ ]
476
+ state = combine_states(states)
477
+ self._record(spec, job_id, state, "")
478
+ else:
479
+ state = self._run_and_record(spec, argv, job_env, job_id)
480
+ state_dir = _local_state_dir()
481
+ where = (
482
+ f"; output in {state_dir / (job_id + '.out')}"
483
+ if state_dir and state != "COMPLETED"
484
+ else ""
485
+ )
486
+ log.info("ran %s locally as job %s: %s%s", spec.job_name, job_id, state, where)
487
+ return job_id
488
+
489
+ def _run_and_record(
490
+ self, spec: JobSpec, argv: list[str], job_env: dict[str, str], job_id: str
491
+ ) -> str:
492
+ try:
493
+ # the job runs in its OWN session (= process group), so the
494
+ # walltime kill takes the whole tree — a job script waiting on
495
+ # children must not leave them running past the walltime, exactly
496
+ # as Slurm kills the job's group
497
+ proc = subprocess.Popen(
498
+ argv,
499
+ stdout=subprocess.PIPE,
500
+ stderr=subprocess.STDOUT,
501
+ text=True,
502
+ start_new_session=True,
503
+ env=job_env,
504
+ )
505
+ except OSError as exc:
506
+ raise SlurmError(f"local job {spec.job_name} failed to start: {exc}") from exc
507
+ try:
508
+ output, _ = proc.communicate(timeout=spec.time_minutes * self.minute_s)
509
+ state = "COMPLETED" if proc.returncode == 0 else "FAILED"
510
+ except subprocess.TimeoutExpired:
511
+ with contextlib.suppress(ProcessLookupError):
512
+ os.killpg(proc.pid, signal.SIGKILL) # pgid == pid (new session)
513
+ try:
514
+ # bounded drain: a child that re-setsid'd ESCAPED the group
515
+ # kill and still holds the pipe — it must not hang the
516
+ # submitter past the walltime. (A cgroup-less backend cannot
517
+ # reach a double-setsid escapee; Slurm's cgroup containment
518
+ # is the real jail — accepted local residual, logged.)
519
+ output, _ = proc.communicate(timeout=10)
520
+ except subprocess.TimeoutExpired:
521
+ if proc.stdout is not None:
522
+ proc.stdout.close()
523
+ output = ""
524
+ log.warning(
525
+ "local job %s: an escaped child survived the walltime kill", spec.job_name
526
+ )
527
+ state = "TIMEOUT"
528
+ self._record(spec, job_id, state, output)
529
+ return state
530
+
531
+ def _record(self, spec: JobSpec, job_id: str, state: str, output: str) -> None:
532
+ state_dir = _local_state_dir()
533
+ if state_dir is not None:
534
+ try:
535
+ state_dir.mkdir(parents=True, exist_ok=True)
536
+ tmp = state_dir / f".{job_id}.{os.getpid()}.tmp"
537
+ tmp.write_text(state)
538
+ os.replace(tmp, state_dir / job_id)
539
+ # the job's combined stdout/stderr beside its state: a failed local
540
+ # job otherwise leaves nothing to read (#295)
541
+ out_fd = os.open(
542
+ state_dir / f"{job_id}.out", os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600
543
+ )
544
+ with os.fdopen(out_fd, "w") as fh:
545
+ fh.write(output)
546
+ # opportunistic prune: one entry per job would leak forever
547
+ # on a long-running loop; anything the sweep could still want
548
+ # is far younger than a day
549
+ cutoff = time.time() - 24 * 3600
550
+ for old in state_dir.iterdir():
551
+ try:
552
+ if old.stat().st_mtime < cutoff:
553
+ old.unlink()
554
+ except OSError:
555
+ pass
556
+ except OSError as exc:
557
+ log.warning("local job %s: state persist failed: %s", spec.job_name, exc)
558
+ if spec.output and spec.output != "/dev/null":
559
+ try:
560
+ with open(spec.output, "a" if "_" in job_id else "w") as fh:
561
+ fh.write(output)
562
+ except OSError as exc:
563
+ log.warning("local job %s: output write failed: %s", spec.job_name, exc)
564
+ self._states[job_id] = state
565
+
566
+ def status(self, job_id: str) -> str:
567
+ _check_job_id(job_id)
568
+ state = self._states.get(job_id, "")
569
+ if state:
570
+ return state
571
+ # another process's job (an attempt's launch, polled by the tick):
572
+ # synchronous jobs are terminal, so the persisted state is the truth
573
+ state_dir = _local_state_dir()
574
+ if state_dir is not None:
575
+ try:
576
+ return (state_dir / job_id).read_text().strip() or GONE
577
+ except OSError:
578
+ pass
579
+ return GONE
580
+
581
+ def pending_reason(self, job_id: str) -> str:
582
+ return "" # synchronous jobs are terminal at submit — never pending
583
+
584
+ def job_partition(self, job_id: str) -> str:
585
+ return "" # no scheduler, no partitions
586
+
587
+ def active_job_names(self) -> list[str]:
588
+ return [] # synchronous: nothing is ever pending or running
589
+
590
+ def queue_snapshot(self) -> list[dict[str, str]]:
591
+ return []
592
+
593
+ def lane_load(self, partition: str) -> dict[str, int]:
594
+ return {} # no lanes in the monolith
595
+
596
+ def job_id_for_name(self, name: str) -> str:
597
+ return ""
598
+
599
+ def cancel(self, job_id: str) -> bool:
600
+ _check_job_id(job_id)
601
+ return True # already terminal; cancelling a finished job is not an error
602
+
603
+
604
+ def parse_elapsed(text: str) -> int | None:
605
+ """Seconds in a sacct Elapsed field: `MM:SS`, `HH:MM:SS` or `D-HH:MM:SS`.
606
+ None for anything else (an unknown field never becomes a refund)."""
607
+ days = 0
608
+ if "-" in text:
609
+ day_part, _, text = text.partition("-")
610
+ if not day_part.isdigit():
611
+ return None
612
+ days = int(day_part)
613
+ parts = text.split(":")
614
+ if not parts or not all(p.isdigit() for p in parts) or len(parts) > 3:
615
+ return None
616
+ nums = [int(p) for p in parts]
617
+ while len(nums) < 3:
618
+ nums.insert(0, 0)
619
+ hours, minutes, seconds = nums
620
+ return days * 86400 + hours * 3600 + minutes * 60 + seconds
621
+
622
+
623
+ def is_terminal(state: str) -> bool:
624
+ """Whether a state string from `status` means the job is over.
625
+
626
+ GONE is deliberately NOT terminal here: it means "no record", and the
627
+ deadline-floor logic decides what that implies — not this predicate.
628
+ """
629
+ return any(state.startswith(t) for t in TERMINAL_STATES)
630
+
631
+
632
+ def is_pending(state: str) -> bool:
633
+ return state.startswith("PENDING")
634
+
635
+
636
+ def gpus_in_gres(gres: str) -> int:
637
+ """GPUs a queue row asks for, from squeue's %b (`gres/gpu:8`,
638
+ `gres/gpu:h200:2`, `gres:gpu:4`; `N/A` or anything else = 0)."""
639
+ total = 0
640
+ for part in gres.replace(";", ",").split(","):
641
+ fields = part.strip().split(":")
642
+ if (len(fields) >= 2 and fields[0] in ("gres/gpu", "gpu")) or (
643
+ len(fields) >= 3 and fields[0] == "gres" and fields[1] == "gpu"
644
+ ):
645
+ try:
646
+ total += int(fields[-1])
647
+ except ValueError:
648
+ continue
649
+ return total
650
+
651
+
652
+ def quote_command(parts: Sequence[str]) -> str:
653
+ """Shell-quote a command for JobSpec.command (--wrap takes a string)."""
654
+ return " ".join(shlex.quote(p) for p in parts)