outerloop-science 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. outerloop/__init__.py +18 -0
  2. outerloop/__main__.py +3 -0
  3. outerloop/appauth.py +230 -0
  4. outerloop/appmanifest.py +203 -0
  5. outerloop/attempt.py +3784 -0
  6. outerloop/brief.py +528 -0
  7. outerloop/cli.py +621 -0
  8. outerloop/climbboard.py +1395 -0
  9. outerloop/compute.py +654 -0
  10. outerloop/contract.py +492 -0
  11. outerloop/contract_cli.py +63 -0
  12. outerloop/disk.py +164 -0
  13. outerloop/dispatch.py +631 -0
  14. outerloop/evalcache.py +147 -0
  15. outerloop/followup.py +2172 -0
  16. outerloop/github.py +1531 -0
  17. outerloop/harness.py +1435 -0
  18. outerloop/housekeeping.py +151 -0
  19. outerloop/image.py +368 -0
  20. outerloop/init.py +744 -0
  21. outerloop/intake.py +126 -0
  22. outerloop/launchlog.py +239 -0
  23. outerloop/limits.py +80 -0
  24. outerloop/maintain.py +353 -0
  25. outerloop/maintain_agent_cli.py +81 -0
  26. outerloop/maintain_post_cli.py +140 -0
  27. outerloop/markers.py +48 -0
  28. outerloop/measure.py +529 -0
  29. outerloop/orchestrator.py +2011 -0
  30. outerloop/panel.py +188 -0
  31. outerloop/paths.py +40 -0
  32. outerloop/posting.py +160 -0
  33. outerloop/progress.py +170 -0
  34. outerloop/py.typed +0 -0
  35. outerloop/review.py +615 -0
  36. outerloop/review_agent.py +263 -0
  37. outerloop/review_agent_cli.py +209 -0
  38. outerloop/review_post_cli.py +162 -0
  39. outerloop/review_summarize_cli.py +165 -0
  40. outerloop/role_runner.py +229 -0
  41. outerloop/roles.py +274 -0
  42. outerloop/rolespec.py +91 -0
  43. outerloop/runstate.py +385 -0
  44. outerloop/steward.py +845 -0
  45. outerloop/style.py +12 -0
  46. outerloop/syscall.py +1192 -0
  47. outerloop/syscall_cli.py +762 -0
  48. outerloop/tick.py +3422 -0
  49. outerloop/verifier.py +403 -0
  50. outerloop/verify_agent.py +151 -0
  51. outerloop/verify_agent_cli.py +95 -0
  52. outerloop/verify_post_cli.py +116 -0
  53. outerloop/watcher.py +203 -0
  54. outerloop_science-0.1.0.dist-info/METADATA +152 -0
  55. outerloop_science-0.1.0.dist-info/RECORD +59 -0
  56. outerloop_science-0.1.0.dist-info/WHEEL +4 -0
  57. outerloop_science-0.1.0.dist-info/entry_points.txt +2 -0
  58. outerloop_science-0.1.0.dist-info/licenses/LICENSE +202 -0
  59. outerloop_science-0.1.0.dist-info/licenses/NOTICE +5 -0
outerloop/measure.py ADDED
@@ -0,0 +1,529 @@
1
+ """Dispatched measurement: run a climb's evals as their own jobs, park, resume.
2
+
3
+ Stage B (part 1) of docs/design/dispatcher.md, built on the `dispatch`
4
+ primitive. The insight that makes a climb resumable across a process death:
5
+ the SESSION cannot be re-run on wake (it already made its edits), but the
6
+ MEASURE-AND-DECIDE phase after the candidate is committed is a pure function
7
+ of committed trees + contract — so every measurement is cacheable by its
8
+ identity and the whole phase re-runs idempotently.
9
+
10
+ A `DispatchedMeasurer` turns a set of `Measure`s (each a committed tree sha +
11
+ the contract command) into: submit every not-yet-done measure as its own
12
+ eval job (one afterany wake covers the set), PARK by raising
13
+ `MeasurementPending`; on the wake, `results()` reads every job's output from
14
+ the run directory — a completed measure returns instantly, so the resumed
15
+ phase flows straight through to the decision.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import hashlib
21
+ import json
22
+ import logging
23
+ from dataclasses import dataclass
24
+ from pathlib import Path
25
+ from typing import Any
26
+
27
+ from outerloop.compute import GONE, Compute, JobSpec, is_terminal
28
+ from outerloop.dispatch import (
29
+ eval_job_spec,
30
+ read_eval_result,
31
+ write_eval_job,
32
+ )
33
+ from outerloop.orchestrator import EvalError
34
+
35
+ log = logging.getLogger(__name__)
36
+
37
+
38
+ class MeasurementPending(Exception):
39
+ """Raised when one or more measures have not completed. Carries the wake
40
+ dependency (the colon-joined job ids: `afterany:<a>:<b>` in one wake job)
41
+ so the caller can park the run as `waiting` on exactly this set."""
42
+
43
+ def __init__(self, job_ids: tuple[str, ...]):
44
+ self.job_ids = job_ids
45
+ super().__init__(f"{len(job_ids)} measure(s) pending: {':'.join(job_ids)}")
46
+
47
+ def afterany(self) -> str:
48
+ """The Slurm dependency for the wake job, or "" when the pending set
49
+ carries no known ids (a transient query failure) — the caller then
50
+ relies on the tick sweep's deadline instead of an afterany wake."""
51
+ return "afterany:" + ":".join(self.job_ids) if self.job_ids else ""
52
+
53
+
54
+ @dataclass(frozen=True)
55
+ class Measure:
56
+ """One eval a climb needs: a COMMITTED tree sha measured by the contract
57
+ command. `name` keys its result file and must be unique within a climb
58
+ (e.g. `baseline`, `candidate`, `sib-tsp-base`). `extra_env` carries the
59
+ paired seed for suite comparisons."""
60
+
61
+ name: str
62
+ tree_sha: str
63
+ command: str
64
+ metric: str
65
+ extra_env: tuple[tuple[str, str], ...] = ()
66
+ # the measured BENCHMARK's GPUs — per measure, because a suite gate
67
+ # measures siblings that may need a different lane than the climbed one
68
+ gpus: int = 0
69
+
70
+ def env(self) -> dict[str, str]:
71
+ return dict(self.extra_env)
72
+
73
+
74
+ @dataclass(frozen=True)
75
+ class SiblingSpec:
76
+ """One suite sibling's resolved measurement facts. Each sibling carries
77
+ its OWN seed variable — a sibling that resamples reads a DIFFERENT env var
78
+ than the climbed benchmark (`sib.seed_env`, not the benchmark's). The
79
+ in-job gate draws ONE `suite_seed` and hands that single value to every
80
+ sibling through its own var, so callers set every sibling's `seed` to that
81
+ one suite_seed; the pair (base and cand) always shares it (common random
82
+ numbers). The per-sibling `seed` field only exists so the var, not the
83
+ value, can differ."""
84
+
85
+ name: str
86
+ command: str
87
+ metric: str
88
+ seed_env: str = ""
89
+ seed: int = 0
90
+ gpus: int = 0
91
+
92
+ def env(self) -> tuple[tuple[str, str], ...]:
93
+ # inject only a REAL drawn seed: 0 is the ledger's "no seed recorded"
94
+ # sentinel (draw_run_seed returns 1+), so a seeded sibling with an
95
+ # unset (0) seed injects no var rather than a literal "0" that would
96
+ # read as a real seed. Guards key off truthiness, per the convention.
97
+ return ((self.seed_env, str(self.seed)),) if self.seed_env and self.seed else ()
98
+
99
+
100
+ def plan_measures(
101
+ command: str,
102
+ metric: str,
103
+ base_sha: str,
104
+ candidate_sha: str,
105
+ seed_env: str = "",
106
+ seed: int = 0,
107
+ siblings: tuple[SiblingSpec, ...] = (),
108
+ gpus: int = 0,
109
+ ) -> list[Measure]:
110
+ """The measures a climb needs, as a pure function of its committed shas
111
+ and contract facts — the same inputs a wake process reconstructs from the
112
+ run record, so the plan is identical before and after a park.
113
+
114
+ Always: `baseline` @ base_sha and `candidate` @ candidate_sha, paired on
115
+ the same `seed` (common random numbers) when the benchmark resamples.
116
+ For a suite gate, each `SiblingSpec` contributes a paired `sib-<name>-base`
117
+ @ base_sha and `sib-<name>-cand` @ candidate_sha — each on the SIBLING's
118
+ OWN seed_env and seed, exactly the 2N-paired comparison the in-job gate
119
+ computes, now dispatched.
120
+ """
121
+ # inject only a REAL drawn seed (>= 1); seed 0 is the "no seed recorded"
122
+ # sentinel, never a value to run under (see SiblingSpec.env / draw_run_seed).
123
+ env: tuple[tuple[str, str], ...] = ((seed_env, str(seed)),) if seed_env and seed else ()
124
+ plan = [
125
+ Measure("baseline", base_sha, command, metric, env, gpus=gpus),
126
+ Measure("candidate", candidate_sha, command, metric, env, gpus=gpus),
127
+ ]
128
+ for sib in siblings:
129
+ plan.append(
130
+ Measure(
131
+ f"sib-{sib.name}-base", base_sha, sib.command, sib.metric, sib.env(), gpus=sib.gpus
132
+ )
133
+ )
134
+ plan.append(
135
+ Measure(
136
+ f"sib-{sib.name}-cand",
137
+ candidate_sha,
138
+ sib.command,
139
+ sib.metric,
140
+ sib.env(),
141
+ gpus=sib.gpus,
142
+ )
143
+ )
144
+ return plan
145
+
146
+
147
+ def _baseline_cache_path(cache_dir: Path, benchmark: str, base_sha: str) -> Path:
148
+ return cache_dir / f"{benchmark}@{base_sha}.json"
149
+
150
+
151
+ def _no_result_note(job_id: str, state: str) -> str:
152
+ """The note for a job that ended with no result. A TIMEOUT is the walltime
153
+ the submit declared (or the contract's default), and the author must
154
+ hear that a slower run needs more minutes — not that measurement broke."""
155
+ if state.startswith("TIMEOUT"):
156
+ return (
157
+ f"job {job_id} hit its walltime (TIMEOUT) before producing a result; "
158
+ "a slower run needs more minutes than were declared for its eval"
159
+ )
160
+ if state and state != GONE and is_terminal(state):
161
+ return f"job {job_id} ended {state} without a result"
162
+ return "dispatched job vanished without a result"
163
+
164
+
165
+ def read_baseline_cache(
166
+ cache_dir: Path,
167
+ benchmark: str,
168
+ base_sha: str,
169
+ *,
170
+ image: str = "",
171
+ command: str = "",
172
+ metric: str = "",
173
+ seed_env: str = "",
174
+ gpus: int = 0,
175
+ ) -> dict[str, Any] | None:
176
+ """The cached base-tree measurement for (benchmark, base sha), or None.
177
+ The entry must have been measured under the SAME determinants the
178
+ candidate will be — eval image, contract command, metric key, seed
179
+ variable, GPU count: everything the measurer's own eval identity
180
+ carries except the tree sha (the key) and the seed VALUE (fresh per
181
+ attempt by design) — or it is stale (terra #178): a comparison across
182
+ determinants is not a comparison. A cache
183
+ entry is only ever written from an orchestrator-measured value (below),
184
+ never from anything an author produced."""
185
+ try:
186
+ data = json.loads(_baseline_cache_path(cache_dir, benchmark, base_sha).read_text())
187
+ except (OSError, ValueError):
188
+ return None
189
+ if not isinstance(data, dict) or "value" not in data:
190
+ return None
191
+ try:
192
+ float(data["value"])
193
+ except (TypeError, ValueError):
194
+ return None
195
+ if (
196
+ data.get("image", "") != image
197
+ or data.get("command", "") != command
198
+ or data.get("metric", "") != metric
199
+ or data.get("seed_env", "") != seed_env
200
+ or int(data.get("gpus", 0) or 0) != gpus
201
+ ):
202
+ return None
203
+ return data
204
+
205
+
206
+ def write_baseline_cache(
207
+ cache_dir: Path,
208
+ benchmark: str,
209
+ base_sha: str,
210
+ *,
211
+ value: float,
212
+ seed: int,
213
+ run_tag: str,
214
+ image: str = "",
215
+ command: str = "",
216
+ metric: str = "",
217
+ seed_env: str = "",
218
+ gpus: int = 0,
219
+ ) -> None:
220
+ """Record an orchestrator-measured baseline for every later attempt on
221
+ this base, with the determinants it was measured under. Atomic (a
222
+ unique tmp per writer + replace): two width slots measuring the same
223
+ base concurrently both land a valid file; last writer wins, and both
224
+ values are real measurements."""
225
+ import os
226
+ import tempfile
227
+
228
+ cache_dir.mkdir(parents=True, exist_ok=True)
229
+ path = _baseline_cache_path(cache_dir, benchmark, base_sha)
230
+ fd, tmp_name = tempfile.mkstemp(prefix=path.name + ".", suffix=".tmp", dir=cache_dir)
231
+ with os.fdopen(fd, "w") as fh:
232
+ json.dump(
233
+ {
234
+ "value": value,
235
+ "seed": seed,
236
+ "run": run_tag,
237
+ "base_sha": base_sha,
238
+ "image": image,
239
+ "command": command,
240
+ "metric": metric,
241
+ "seed_env": seed_env,
242
+ "gpus": gpus,
243
+ },
244
+ fh,
245
+ )
246
+ Path(tmp_name).replace(path)
247
+
248
+
249
+ @dataclass
250
+ class DispatchedMeasurer:
251
+ """Submits and reads a climb's measures as jobs on any `Compute` backend.
252
+ Stateless beyond the compute handle: all durable state is the eval job
253
+ output in the run dir, so a fresh process on wake reads exactly what the
254
+ parked one submitted. On a synchronous backend (`LocalCompute`) every job
255
+ is done by the time it is checked, so nothing parks and `results()` flows
256
+ straight through — the one measurer covers dispatched and local evals."""
257
+
258
+ compute: Compute
259
+ run_dir: Path
260
+ repo_root: Path
261
+ image: str
262
+ account: str
263
+ partition: str
264
+ eval_minutes: int
265
+ run_tag: str = "run" # disambiguates job names across runs on one account
266
+ # the GPU lane; a measure with `gpus > 0` is placed there at dispatch
267
+ # time (per MEASURE — a suite's siblings may differ from the climbed
268
+ # benchmark), everything else on account/partition
269
+ gpu_partition: str = ""
270
+ gpu_account: str = ""
271
+ # where a `baseline: cached` benchmark's base-tree measurements live
272
+ # (target-wide); None = no cache, every gate measures its own baseline
273
+ baseline_cache: Path | None = None
274
+ # the target's kernel-warmed seed cache, copied into each job (evalcache)
275
+ seed_cache: Path | None = None
276
+
277
+ def _placement(self, m: Measure) -> tuple[str, str]:
278
+ if m.gpus <= 0:
279
+ return self.account, self.partition
280
+ if not self.gpu_partition:
281
+ raise ValueError(
282
+ f"measure {m.name} needs {m.gpus} GPU(s) but no GPU lane is configured "
283
+ "(set OUTERLOOP_GPU_PARTITION)"
284
+ )
285
+ return self.gpu_account or self.account, self.gpu_partition
286
+
287
+ def _det(self, m: Measure) -> str:
288
+ # Everything a measure's RESULT depends on and that can vary across a
289
+ # PARK/RESUME (when a fresh measurer reads this run_dir): the container
290
+ # image, the measure's logical role, the code (tree_sha), and the
291
+ # contract facts it is evaluated under (command, metric, seeded env).
292
+ # A cache key missing any of these would return a value computed under
293
+ # DIFFERENT inputs — e.g. a resume that re-fetched the contract after
294
+ # its command changed, or ran under a rebuilt image, reading the stale
295
+ # pre-change result. (account / walltime don't change a result's value,
296
+ # only whether it completes.) NUL separators keep the parts unambiguous
297
+ # (`a`+`bc` != `ab`+`c`).
298
+ env = "".join(f"\0{k}={v}" for k, v in sorted(m.env().items()))
299
+ return f"{self.image}\0{m.name}\0{m.tree_sha}\0{m.command}\0{m.metric}{env}"
300
+
301
+ def _slot(self, m: Measure) -> str:
302
+ # Storage identity = the full determinant, with NOTHING truncated: the
303
+ # eval dir is the durable result cache, so any prefix could alias two
304
+ # distinct measurements into one stale read. Readable role + FULL sha
305
+ # (verbatim, debuggable) + the FULL sha1 hex of the whole determinant
306
+ # (the collision-free disambiguator for the contract facts the sha
307
+ # alone does not pin — command/metric/seed). A re-measure that changes
308
+ # the sha OR any contract input lands in a fresh dir; a resume with
309
+ # identical inputs reuses it. `m.name` stays the caller-facing key
310
+ # (results["candidate"]). A dir has 255 chars to spare (~91 used).
311
+ h = hashlib.sha1(self._det(m).encode()).hexdigest()
312
+ return f"{m.name}-{m.tree_sha}-{h}"
313
+
314
+ def _ev(self, m: Measure) -> Path:
315
+ return self.run_dir / f"eval-{self._slot(m)}"
316
+
317
+ def _job_name(self, m: Measure) -> str:
318
+ # A LIVENESS HINT, not a durable key: the cluster is asked "is this
319
+ # measure's job live" by name. Slurm caps name length, so this hash is
320
+ # necessarily bounded (16 hex = 64 bits) rather than full-width like the
321
+ # slot. That bound is safe because a job-name collision cannot cause a
322
+ # stale RESULT — results are read from the collision-free slot; the
323
+ # worst case is one measure seeing another's job as "live" and parking
324
+ # instead of dispatching, which the deadline sweep then re-checks. The
325
+ # hash covers the whole determinant (+ run_tag, which disambiguates
326
+ # jobs across runs sharing one Slurm account); the readable prefixes
327
+ # are for a human reading squeue.
328
+ h = hashlib.sha1(f"{self.run_tag}\0{self._det(m)}".encode()).hexdigest()[:16]
329
+ return f"eval-{self.run_tag[:10]}-{m.name[:12]}-{h}"
330
+
331
+ def _done(self, m: Measure) -> bool:
332
+ return (self._ev(m) / "exit-code").exists()
333
+
334
+ def _ended_without_result(self, m: Measure) -> str:
335
+ """Why a dispatched job produced no result; a TIMEOUT means the eval
336
+ needs more walltime."""
337
+ job_id = self._marker(m)
338
+ try:
339
+ state = self.compute.status(job_id) if job_id.isdigit() else ""
340
+ except Exception:
341
+ state = ""
342
+ return _no_result_note(job_id, state)
343
+
344
+ def _marker(self, m: Measure) -> str:
345
+ f = self._ev(m) / "submitted"
346
+ return f.read_text().strip() if f.exists() else ""
347
+
348
+ def _dispatch(self, m: Measure) -> str:
349
+ script = write_eval_job(
350
+ self.run_dir,
351
+ self._slot(m),
352
+ repo_root=self.repo_root,
353
+ snapshot_sha=m.tree_sha,
354
+ command=m.command,
355
+ image=self.image,
356
+ extra_env=m.env(),
357
+ gpus=m.gpus,
358
+ seed_cache=self.seed_cache,
359
+ )
360
+ account, partition = self._placement(m)
361
+ spec: JobSpec = eval_job_spec(
362
+ script,
363
+ job_name=self._job_name(m),
364
+ account=account,
365
+ partition=partition,
366
+ eval_minutes=self.eval_minutes,
367
+ gpus=m.gpus,
368
+ )
369
+ job_id = self.compute.submit(spec)
370
+ (self._ev(m) / "submitted").write_text(job_id)
371
+ log.info("dispatched measure %s (sha %s) as job %s", m.name, m.tree_sha[:12], job_id)
372
+ return job_id
373
+
374
+ def results(self, measures: list[Measure]) -> dict[str, float]:
375
+ """Every measure's value, or PARK / FAIL. The CLUSTER is the source of
376
+ truth for liveness (a job named for the measure, found by squeue),
377
+ never just the local marker — so a submitter that died before writing
378
+ its id cannot cause a duplicate submit. Per not-yet-done measure:
379
+ * a live job with this name -> park on its id (authoritative);
380
+ * a marker but NO live job -> the job ran and vanished without a
381
+ result (SIGKILL / node death / GONE) -> EvalError, never resubmit;
382
+ * no live job and no marker -> never dispatched -> dispatch;
383
+ * squeue unavailable -> park (marker id if any) rather than risk a
384
+ duplicate; the wake set may be empty -> the sweep deadline retries.
385
+ """
386
+ pending: list[str] = []
387
+ blind = False
388
+ for m in measures:
389
+ if self._done(m):
390
+ continue
391
+ try:
392
+ live = self.compute.job_id_for_name(self._job_name(m))
393
+ except Exception:
394
+ # cannot tell if it is running: do NOT dispatch (would risk a
395
+ # duplicate) and do NOT declare it dead — re-check next wake
396
+ marker = self._marker(m)
397
+ if marker:
398
+ pending.append(marker)
399
+ else:
400
+ blind = True
401
+ continue
402
+ if live:
403
+ pending.append(live) # queued or running — the real job id
404
+ continue
405
+ if self._marker(m):
406
+ # was dispatched, not live, no result -> died before result
407
+ raise EvalError(f"measure {m.name}: {self._ended_without_result(m)}")
408
+ # No marker, not live, no result -> never dispatched -> dispatch.
409
+ # RESIDUAL (bounded, accepted): if a prior process died in the
410
+ # microsecond gap between sbatch returning and _dispatch writing
411
+ # the marker, AND that orphaned job then ran and died without a
412
+ # result, its name is gone from squeue and this redispatches once
413
+ # (never loops — the redispatch writes a marker). Cost is one
414
+ # wasted eval in a triple-failure conjunction; fully closing it
415
+ # needs sacct-by-name over job history, not worth that surface.
416
+ job_id = self._dispatch(m)
417
+ if self._done(m):
418
+ continue # a synchronous compute finished the job inside submit
419
+ try:
420
+ state = self.compute.status(job_id)
421
+ except Exception:
422
+ state = "" # status unknown right after submit is normal; park
423
+ if state and is_terminal(state):
424
+ # the job already ENDED without writing a result (a local
425
+ # timeout, an instant cluster failure): parking would wait on
426
+ # a job that will never deliver — fail like a vanished job
427
+ raise EvalError(f"measure {m.name}: {_no_result_note(job_id, state)}")
428
+ pending.append(job_id)
429
+ if pending or blind:
430
+ raise MeasurementPending(tuple(pending))
431
+ out: dict[str, float] = {}
432
+ for m in measures:
433
+ try:
434
+ out[m.name] = read_eval_result(self.run_dir, self._slot(m), m.metric)
435
+ except EvalError as exc:
436
+ raise EvalError(self._explain_signal_exit(m, str(exc))) from None
437
+ return out
438
+
439
+ def _explain_signal_exit(self, m: Measure, message: str) -> str:
440
+ """Exit 143 is the job script's record of a TERM: Slurm sends one at
441
+ the walltime, on scancel, and on preemption alike, so only its end
442
+ state for the job says which. The walltime case is the author's to
443
+ fix (more minutes); the others are the cluster's."""
444
+ try:
445
+ code = (self._ev(m) / "exit-code").read_text().strip()
446
+ except OSError:
447
+ return message
448
+ if code != "143":
449
+ return message
450
+ job_id = self._marker(m)
451
+ try:
452
+ state = self.compute.status(job_id) if job_id.isdigit() else ""
453
+ except Exception:
454
+ state = ""
455
+ if state.startswith("TIMEOUT"):
456
+ why = (
457
+ "was killed at its walltime (exit 143); a slower run needs more minutes "
458
+ "than were declared for its eval"
459
+ )
460
+ elif state.startswith("CANCELLED"):
461
+ why = "was cancelled (exit 143)"
462
+ elif state.startswith("PREEMPTED"):
463
+ why = "was preempted (exit 143); nothing about the tree is known"
464
+ else:
465
+ why = "was killed by a signal (exit 143)"
466
+ return f"measure {m.name} {why}; {message}"
467
+
468
+
469
+ @dataclass(frozen=True)
470
+ class DispatchSettings:
471
+ """The cluster coordinates a dispatched measurer needs, grouped so the
472
+ composition root (the climb CLI) reads them ONCE from its args/env and the
473
+ climb just carries them. The per-run pieces (run dir, snapshot repo, the
474
+ benchmark's eval hint, the run tag) are bound at build time by `measurer`,
475
+ so this stays a static description of WHERE to dispatch, not a live handle
476
+ to one run."""
477
+
478
+ compute: Compute
479
+ image: str
480
+ account: str
481
+ partition: str
482
+ # the GPU lane: where jobs of a benchmark with `gpus > 0` go. Empty
483
+ # gpu_partition = this deployment cannot place GPU jobs (the tick refuses
484
+ # to launch such benchmarks); empty gpu_account = same account as CPU jobs.
485
+ gpu_partition: str = ""
486
+ gpu_account: str = ""
487
+ # the target's seed cache (evalcache.seed_dir), copied into every job
488
+ seed_cache: Path | None = None
489
+
490
+ def placement(self, gpus: int) -> tuple[str, str]:
491
+ """(account, partition) for a job needing `gpus` GPUs. Raises when a
492
+ GPU job has no lane — a queue that can never run is worse than a
493
+ loud refusal. Local compute has no lanes: jobs are subprocesses on
494
+ whatever GPUs the machine has, so placement is empty by design."""
495
+ from outerloop.compute import local_mode
496
+
497
+ if gpus <= 0 or local_mode():
498
+ return self.account, self.partition
499
+ if not self.gpu_partition:
500
+ raise ValueError(
501
+ f"benchmark needs {gpus} GPU(s) but no GPU lane is configured "
502
+ "(set OUTERLOOP_GPU_PARTITION)"
503
+ )
504
+ return self.gpu_account or self.account, self.gpu_partition
505
+
506
+ def measurer(
507
+ self, run_dir: Path, repo_root: Path, eval_minutes: int, run_tag: str
508
+ ) -> DispatchedMeasurer:
509
+ """Bind these coordinates to one run's dispatched measurer. `repo_root`
510
+ is the workspace whose `refs/dispatch/*` snapshots the eval jobs check
511
+ out; `eval_minutes` is the benchmark's contract hint (clamped in the
512
+ job spec). GPUs are per MEASURE (Measure.gpus): the measurer carries
513
+ the lane and places each measure when it dispatches it."""
514
+ return DispatchedMeasurer(
515
+ compute=self.compute,
516
+ run_dir=run_dir,
517
+ repo_root=repo_root,
518
+ image=self.image,
519
+ account=self.account,
520
+ partition=self.partition,
521
+ eval_minutes=eval_minutes,
522
+ run_tag=run_tag,
523
+ gpu_partition=self.gpu_partition,
524
+ gpu_account=self.gpu_account,
525
+ # target-wide, beside the run dirs: every attempt on one base
526
+ # shares its cached baseline measurement (Benchmark.baseline)
527
+ baseline_cache=run_dir.parent / "baselines",
528
+ seed_cache=self.seed_cache,
529
+ )