outerloop-science 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. outerloop/__init__.py +18 -0
  2. outerloop/__main__.py +3 -0
  3. outerloop/appauth.py +230 -0
  4. outerloop/appmanifest.py +203 -0
  5. outerloop/attempt.py +3784 -0
  6. outerloop/brief.py +528 -0
  7. outerloop/cli.py +621 -0
  8. outerloop/climbboard.py +1395 -0
  9. outerloop/compute.py +654 -0
  10. outerloop/contract.py +492 -0
  11. outerloop/contract_cli.py +63 -0
  12. outerloop/disk.py +164 -0
  13. outerloop/dispatch.py +631 -0
  14. outerloop/evalcache.py +147 -0
  15. outerloop/followup.py +2172 -0
  16. outerloop/github.py +1531 -0
  17. outerloop/harness.py +1435 -0
  18. outerloop/housekeeping.py +151 -0
  19. outerloop/image.py +368 -0
  20. outerloop/init.py +744 -0
  21. outerloop/intake.py +126 -0
  22. outerloop/launchlog.py +239 -0
  23. outerloop/limits.py +80 -0
  24. outerloop/maintain.py +353 -0
  25. outerloop/maintain_agent_cli.py +81 -0
  26. outerloop/maintain_post_cli.py +140 -0
  27. outerloop/markers.py +48 -0
  28. outerloop/measure.py +529 -0
  29. outerloop/orchestrator.py +2011 -0
  30. outerloop/panel.py +188 -0
  31. outerloop/paths.py +40 -0
  32. outerloop/posting.py +160 -0
  33. outerloop/progress.py +170 -0
  34. outerloop/py.typed +0 -0
  35. outerloop/review.py +615 -0
  36. outerloop/review_agent.py +263 -0
  37. outerloop/review_agent_cli.py +209 -0
  38. outerloop/review_post_cli.py +162 -0
  39. outerloop/review_summarize_cli.py +165 -0
  40. outerloop/role_runner.py +229 -0
  41. outerloop/roles.py +274 -0
  42. outerloop/rolespec.py +91 -0
  43. outerloop/runstate.py +385 -0
  44. outerloop/steward.py +845 -0
  45. outerloop/style.py +12 -0
  46. outerloop/syscall.py +1192 -0
  47. outerloop/syscall_cli.py +762 -0
  48. outerloop/tick.py +3422 -0
  49. outerloop/verifier.py +403 -0
  50. outerloop/verify_agent.py +151 -0
  51. outerloop/verify_agent_cli.py +95 -0
  52. outerloop/verify_post_cli.py +116 -0
  53. outerloop/watcher.py +203 -0
  54. outerloop_science-0.1.0.dist-info/METADATA +152 -0
  55. outerloop_science-0.1.0.dist-info/RECORD +59 -0
  56. outerloop_science-0.1.0.dist-info/WHEEL +4 -0
  57. outerloop_science-0.1.0.dist-info/entry_points.txt +2 -0
  58. outerloop_science-0.1.0.dist-info/licenses/LICENSE +202 -0
  59. outerloop_science-0.1.0.dist-info/licenses/NOTICE +5 -0
outerloop/contract.py ADDED
@@ -0,0 +1,492 @@
1
+ """Contract schema and loader for a target repo's contract file (`.outerloop.yaml`).
2
+
3
+ The contract is the opt-in declaration: benchmarks, budgets, scope. The loader
4
+ enforces invariants no YAML can override (see the threat model in
5
+ docs/design/architecture.md): autoresearch is never a target of itself, and the
6
+ contract file, the target's roadmap, and `.github/` are always forbidden write
7
+ paths, regardless of what `scope.allowed` says.
8
+
9
+ Contracts live in target repos and are therefore untrusted input: parsing
10
+ rejects aliases, duplicate keys, and oversized documents.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import posixpath
16
+ import re
17
+ from collections.abc import Callable
18
+ from pathlib import Path, PurePosixPath
19
+ from typing import Any, Literal
20
+
21
+ import yaml
22
+ from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator
23
+
24
+ SELF_REPO = "outerloop-science/outerloop"
25
+ # The contract's filename in the target repo. New adopters write `.outerloop.yaml`;
26
+ # `.autoresearch.yaml` (targets written before the rename) is still honored. Every
27
+ # read goes through `find_contract`, which tries the new name first. Neither is
28
+ # ever a writable path for the agent.
29
+ # The pre-rename name is dropped in the release after 0.1.
30
+ CONTRACT_NAMES: tuple[str, ...] = (".outerloop.yaml", ".autoresearch.yaml")
31
+ CONTRACT_NAME = CONTRACT_NAMES[0] # what the docs and new contracts use
32
+ ALWAYS_FORBIDDEN: tuple[str, ...] = (".github", *CONTRACT_NAMES)
33
+ MAX_CONTRACT_BYTES = 64 * 1024
34
+ _GLOB_CHARS = set("*?[]!")
35
+
36
+
37
+ def find_contract(read: Callable[[str], str | None]) -> tuple[str, str] | None:
38
+ """(name, text) of the first contract file `read` yields, new name first; None
39
+ when the target has neither. `read(name)` returns the file's text, or None when
40
+ that name is absent — wrap a reader that raises instead."""
41
+ for name in CONTRACT_NAMES:
42
+ text = read(name)
43
+ if text is not None:
44
+ return name, text
45
+ return None
46
+
47
+
48
+ def contract_in_tree(tree: Path) -> tuple[str, str] | None:
49
+ """The contract in a checked-out tree: (name, text), or None."""
50
+ return find_contract(lambda n: (tree / n).read_text() if (tree / n).is_file() else None)
51
+
52
+
53
+ def contract_text_in_tree(tree: Path) -> str:
54
+ """The contract's text in a checked-out tree; raises FileNotFoundError (as a
55
+ direct read did) naming both candidates when the tree has neither."""
56
+ found = contract_in_tree(tree)
57
+ if found is None:
58
+ raise FileNotFoundError(f"no contract in {tree} ({' or '.join(CONTRACT_NAMES)})")
59
+ return found[1]
60
+
61
+
62
+ class ContractError(ValueError):
63
+ """Base class for contract rejections."""
64
+
65
+
66
+ class SelfTargetError(ContractError):
67
+ """Raised when a contract names autoresearch itself as the target."""
68
+
69
+
70
+ class ScopeError(ContractError):
71
+ """Raised when an allowed path is unsafe or overlaps a forbidden path."""
72
+
73
+
74
+ class _SafeLoader(yaml.SafeLoader):
75
+ """SafeLoader that refuses alias expansion and duplicate mapping keys."""
76
+
77
+ def compose_node(self, parent: Any, index: Any) -> Any:
78
+ if self.check_event(yaml.events.AliasEvent):
79
+ raise ContractError("YAML aliases are not allowed in contracts")
80
+ return super().compose_node(parent, index)
81
+
82
+ def construct_mapping(self, node: Any, deep: bool = False) -> dict[Any, Any]:
83
+ seen = set()
84
+ for key_node, _ in node.value:
85
+ key = self.construct_object(key_node, deep=deep)
86
+ if key in seen:
87
+ raise ContractError(f"duplicate key in contract: {key!r}")
88
+ seen.add(key)
89
+ return super().construct_mapping(node, deep=deep)
90
+
91
+
92
+ class _StrictModel(BaseModel):
93
+ # Typos in a contract must fail loudly, never be silently ignored.
94
+ model_config = ConfigDict(extra="forbid")
95
+
96
+
97
+ class Benchmark(_StrictModel):
98
+ # Slug shape only: the name reaches branch names, ledger keys, and log
99
+ # labels — contract text must not shape refs or paths beyond a slug.
100
+ name: str = Field(min_length=1, pattern=r"^[A-Za-z0-9_.-]{1,64}$")
101
+ command: str = Field(min_length=1)
102
+ metric: str = Field(min_length=1)
103
+ direction: Literal["min", "max"]
104
+ # Significant digits for HUMAN surfaces (PR tables, BENCHMARKS.md,
105
+ # replies) per the benchmark's community convention. Full precision
106
+ # lives only in results/leader.json, the machine ledger (maintainer
107
+ # decision 2026-08-09). Display-only: comparisons always use full
108
+ # floats, so this can never hide or fake an improvement.
109
+ display_digits: int | None = Field(default=None, ge=2, le=12)
110
+ # Resampled-pool benchmarks: the env var the eval reads its run seed
111
+ # from (e.g. PILOT_REACH_SEED). When set, the orchestrator draws ONE
112
+ # fresh seed per measurement pass and pins BOTH sides of a comparison
113
+ # to it (paired, common random numbers), then records it in the ledger
114
+ # row — the number becomes re-derivable instead of pool luck. Strict
115
+ # env-var shape: this string reaches a subprocess environment. RULER
116
+ # INVARIANT the eval must uphold: emit the seed to stdout only, never
117
+ # persist it into the tree — a seed artifact in the workspace would be
118
+ # readable by the solver session that runs between the paired evals
119
+ # (the verifier's ruler read covers this).
120
+ seed_env: str | None = Field(default=None, pattern=r"^[A-Z][A-Z0-9_]{0,63}$")
121
+ # Expected eval duration, minutes — a HINT that turns on dispatched
122
+ # evals (docs/design/dispatcher.md): above the in-job threshold the
123
+ # orchestrator runs this benchmark's evals as their own jobs. Clamped
124
+ # by dispatch.EVAL_JOB_MINUTES_CEILING (our spend cap); it governs only
125
+ # the first eval once measured durations exist. None = in-job (today's
126
+ # behavior).
127
+ eval_minutes: int | None = Field(default=None, ge=1)
128
+ # GPUs for every dispatched job of this benchmark — gate measures and
129
+ # author launches alike. 0 (default) = CPU. A GPU benchmark needs the
130
+ # deployment to name a GPU lane (OUTERLOOP_GPU_PARTITION, optionally
131
+ # OUTERLOOP_GPU_ACCOUNT); without one the tick refuses to launch
132
+ # attempts on it rather than queue evals that can never run. Bounded at
133
+ # one node's worth: multi-node evals are not a shape the jail supports.
134
+ gpus: int = Field(default=0, ge=0, le=8)
135
+ # How the gate obtains the baseline number it compares a candidate to.
136
+ # paired (default): re-measure the base tree next to every candidate,
137
+ # both under one fresh seed (common random numbers) — the noise-
138
+ # minimal comparison, at two evals per attempt.
139
+ # cached: measure the base tree ONCE per (benchmark, base sha) into a
140
+ # cache shared by every attempt on that base, then run only the
141
+ # candidate at a fresh seed. Halves eval spend; the comparison is
142
+ # unpaired, so the contract's min_delta must cover cross-seed noise
143
+ # on its own (calibrate it). The ledger row says which baseline
144
+ # measurement (and seed) a credited delta was taken against.
145
+ baseline: Literal["paired", "cached"] = "paired"
146
+ # Depth budget (docs/design/research-loop.md, "one syscall, author-directed"):
147
+ # how many external experiment jobs the author may LAUNCH within one attempt.
148
+ # A generous meter on actions, not a loop the kernel drives — the author
149
+ # decides what each launch is for. Per-benchmark so "does depth pay here?"
150
+ # is answerable per benchmark. The syscalls are CONTRACT-DRIVEN and on by
151
+ # default wherever the deployment can deliver them (dispatch coords + a
152
+ # resumable backend); `depth_k: 0` is a benchmark's opt-out — the tool is
153
+ # then not offered at all. Weekly spend stays bounded by `runs_per_week`.
154
+ depth_k: int = Field(default=10, ge=0, le=16)
155
+ # Sleep budget, the sibling knob: how many dispatch->sleep->wake cycles the
156
+ # author may spend. Independent of depth_k — launches meter external compute,
157
+ # sleeps meter wake cycles (without this an author could checkpoint-refresh
158
+ # its session clock forever and never launch). Batching is rewarded: many
159
+ # launches under one sleep burn one sleep; a `submit` also rides one sleep.
160
+ sleep_k: int = Field(default=20, ge=1, le=32)
161
+ # Research lines (docs/design/research-lines.md): each agent slot works on
162
+ # its own persistent branch `agents/<agent-id>` — checked out at run start
163
+ # with the base branch merged in and instruction-bearing files reset to the
164
+ # base branch's reviewed versions. Off (default) = today's fork-main-only
165
+ # behavior; the branch substrate is inert until a contract opts in.
166
+ lines: bool = False
167
+
168
+ # Pure loop-steering dials: how often/deep the fleet iterates and how a
169
+ # number renders. Everything else — the command, metric, seed, GPUs,
170
+ # walltime (it selects the execution route), direction, floors, and
171
+ # baseline protocol — defines the measurement or the claim's meaning.
172
+ _WORKFLOW_DIALS = frozenset({"lines", "depth_k", "sleep_k", "display_digits"})
173
+
174
+ def measurement_signature(self) -> tuple:
175
+ """The fields that determine how this benchmark is measured and what
176
+ a claim about it means — every field EXCEPT the pure workflow dials,
177
+ so a future field joins the signature by default and the base-sync
178
+ skip fails toward re-measuring."""
179
+ data = self.model_dump()
180
+ return tuple(sorted((k, repr(v)) for k, v in data.items() if k not in self._WORKFLOW_DIALS))
181
+
182
+ @field_validator("seed_env")
183
+ @classmethod
184
+ def _seed_env_never_managed(cls, value: str | None) -> str | None:
185
+ from outerloop.orchestrator import managed_eval_env
186
+
187
+ if value is not None and managed_eval_env(value):
188
+ raise ValueError(
189
+ f"seed_env must not name or prefix the evaluator's managed environment ({value!r})"
190
+ )
191
+ return value
192
+
193
+ @model_validator(mode="after")
194
+ def _gpu_benchmarks_dispatch(self) -> Benchmark:
195
+ # GPUs only exist on dispatched jobs: the in-job evaluator runs inside
196
+ # the CPU climb job with no allocation and no --nv, so a GPU benchmark
197
+ # under the in-job threshold would measure on a machine with no GPU
198
+ # (terra #174 r2). Make the contract say so up front.
199
+ from outerloop.dispatch import should_dispatch
200
+
201
+ if self.gpus > 0 and not should_dispatch(self.eval_minutes):
202
+ raise ValueError(
203
+ f"benchmark {self.name!r} asks for {self.gpus} GPU(s) but its evals "
204
+ "would run in-job: set eval_minutes above the in-job threshold so they dispatch"
205
+ )
206
+ floor_set = (self.min_delta or 0) > 0 or (self.min_delta_rel or 0) > 0
207
+ if self.baseline == "cached" and not floor_set:
208
+ # an unpaired comparison has no built-in noise cancellation: the
209
+ # floor is the only thing standing between seed luck and a record
210
+ # — and benchmark_floor treats 0 as "no floor", so it must be > 0
211
+ raise ValueError(
212
+ f"benchmark {self.name!r} uses a cached baseline (unpaired comparison) "
213
+ "but declares no positive significance floor: set min_delta or "
214
+ "min_delta_rel (> 0) from a seed-variance calibration"
215
+ )
216
+ return self
217
+
218
+ # Cross-seed noise floor. A comparison against the RECORDED best was
219
+ # measured under a different seed, so a delta inside the floor is noise,
220
+ # not progress; same-seed paired comparisons are exempt by construction.
221
+ # Two forms. min_delta is absolute metric units, right for a bounded
222
+ # metric (a success rate). min_delta_rel is a fraction of the recorded
223
+ # level, right for an unbounded metric (wall-clock timing) whose level
224
+ # drifts with hardware, so an absolute number goes stale. Set either or
225
+ # both; both means the more conservative of the two applies.
226
+ min_delta: float | None = Field(default=None, ge=0)
227
+ # capped at 1.0: a noise floor is a small fraction (e.g. 0.13), so a
228
+ # value above the level itself is a typo (13 meaning 13%), and a floor
229
+ # of 13x would freeze the benchmark. A relative floor assumes a
230
+ # non-zero level; pair it with a small absolute min_delta as a backstop
231
+ # for a metric that can reach 0.
232
+ min_delta_rel: float | None = Field(default=None, ge=0, le=1.0)
233
+
234
+
235
+ class SuiteAggregate(_StrictModel):
236
+ """Unified-benchmark targets: one change is evaluated on every benchmark
237
+ and reported per-env plus this aggregate (no cherry-picking)."""
238
+
239
+ metric: str = Field(min_length=1)
240
+ direction: Literal["min", "max"]
241
+
242
+
243
+ class Budgets(_StrictModel):
244
+ # The attempt's compute allowance, METERED at the syscall for GPU
245
+ # benchmarks: every author launch (minutes x gpus) and a submit's two
246
+ # paired gate evals (2 x eval walltime x gpus) draw on it, and an
247
+ # over-budget request is refused with the numbers. The author may
248
+ # declare its own eval walltime at submit (`submit --minutes`), so a
249
+ # candidate whose eval runs longer is paid for out of this budget rather
250
+ # than killed by a fixed limit — walltime is never the metric; compute
251
+ # is priced here. 0 for CPU benchmarks (nothing to meter).
252
+ gpu_hours_per_run: float = Field(ge=0)
253
+ runs_per_week: int = Field(gt=0)
254
+ # Optional per-repo shaping of the orchestrator's session/job limits.
255
+ # These are WISHES, not grants: limits.effective_limits clamps every
256
+ # value into orchestrator-side [floor, ceiling] bounds, so a target can
257
+ # spend less of us, never more. Absent = orchestrator defaults.
258
+ session_max_turns: int | None = Field(default=None, gt=0)
259
+ session_minutes: int | None = Field(default=None, gt=0)
260
+ attempt_job_minutes: int | None = Field(default=None, gt=0)
261
+ followup_job_minutes: int | None = Field(default=None, gt=0)
262
+ # Per-benchmark self-initiated cooldown, in minutes. Default (unset) is
263
+ # the orchestrator's 6h — right for a standard research repo. An RSI /
264
+ # hot-loop target sets 0: the loop re-dispatches back-to-back and
265
+ # `runs_per_week` becomes the spend guard (Mengye: "for the RSI
266
+ # experiment — no cooldown; make sure the cluster is hot"). Still
267
+ # SERIAL per target until the width dial lands.
268
+ attempt_cooldown_minutes: int | None = Field(default=None, ge=0)
269
+ # THE WIDTH DIAL: concurrent self-initiated attempts per target. Default
270
+ # (unset) = 1, today's serial behavior. A hot-loop target raises it
271
+ # to run N authors abreast — each slot gets its own agent identity
272
+ # (agent-01..agent-0N), so branches, ledger rows, and reports stay
273
+ # distinct. runs_per_week and gpu_hours_per_run remain the spend guards.
274
+ max_active_attempts: int | None = Field(default=None, ge=1)
275
+ # The pace ceiling for an author's sweeps, in GPUs: one launch may hold at
276
+ # most this many at once, so a sweep of N tasks runs
277
+ # max_concurrent_gpus // gpus of them at a time (`--array=0-N%K`). Lenient
278
+ # by design — not the cap divided by the agent count (agents rarely launch
279
+ # at the same moment, and an idle share is wasted GPU); on a 16-GPU cap 12
280
+ # lets one sweep use most of the machine while a sibling's job still gets
281
+ # in. Unset: the author's own pace, the whole array by default.
282
+ max_concurrent_gpus: int | None = Field(default=None, ge=1)
283
+
284
+ @model_validator(mode="before")
285
+ @classmethod
286
+ def _accept_legacy_climb_job_minutes(cls, data: Any) -> Any:
287
+ # TRANSITIONAL: the field was `climb_job_minutes`. Map the legacy key
288
+ # to the new name before validation (new name wins if BOTH appear, so
289
+ # a mid-migration contract never fails), and consume it so extra=forbid
290
+ # does not reject it. Drop this once the (two) live contracts migrate.
291
+ if isinstance(data, dict) and "climb_job_minutes" in data:
292
+ data = dict(data)
293
+ legacy = data.pop("climb_job_minutes")
294
+ data.setdefault("attempt_job_minutes", legacy)
295
+ return data
296
+
297
+
298
+ class Scope(_StrictModel):
299
+ allowed: list[str] = Field(min_length=1)
300
+ # Shared code paths (encoder / world model / training loop — code every
301
+ # benchmark exercises, as opposed to env-specific solver code). A solver
302
+ # diff touching any of these is suite-gated: the orchestrator re-measures
303
+ # EVERY sibling benchmark on both sides and refuses credit if one
304
+ # regresses beyond its own floor. Env-specific diffs stay cheap — only
305
+ # their benchmark is measured.
306
+ shared: list[str] = Field(default_factory=list)
307
+
308
+
309
+ # The orchestrator's ledger: no agent scope — solver OR steward — may
310
+ # contain these; their numbers carry orchestrator provenance only.
311
+ RECORD_PATHS = ("BENCHMARKS.md", "results/leader.json")
312
+
313
+
314
+ class StewardScope(_StrictModel):
315
+ """Paths the BENCHMARK STEWARD may edit: env generators, the eval
316
+ harness, tests, and reference data — NEVER the record ledger
317
+ (BENCHMARKS.md, results/leader.json; the orchestrator writes those
318
+ with its own measurements, and load_contract rejects a steward scope
319
+ that includes them). The solver's `scope.allowed` is implicitly
320
+ forbidden to the steward — the roles' territories must not overlap
321
+ (collusion structure, design/meta.md) — and the always-forbidden set
322
+ (this contract, `.github/`, the roadmap) binds the steward too."""
323
+
324
+ allowed: list[str] = Field(min_length=1)
325
+
326
+
327
+ class Contract(_StrictModel):
328
+ benchmarks: list[Benchmark] = Field(min_length=1)
329
+ budgets: Budgets
330
+ scope: Scope
331
+ roadmap: str = Field(min_length=1)
332
+ suite: SuiteAggregate | None = None
333
+ steward: StewardScope | None = None
334
+ # MERGE POLICY — the autonomy mode dial (docs/design/headline.md), the
335
+ # target owner's declaration like a harness permission mode:
336
+ # manual (default): the bot opens PRs and arms auto-merge only when a
337
+ # required human review stands between arming and merging.
338
+ # auto: a gate+panel-clean PR merges itself. Repo prerequisites the
339
+ # owner sets alongside this knob: "Allow auto-merge" on; branch
340
+ # protection whose required checks are the repo's own CI with
341
+ # STRICT up-to-date enforcement (the branch must match the base —
342
+ # this is what closes the race where the base moves between our
343
+ # freshness check and a direct merge: GitHub itself refuses a
344
+ # stale-base merge); NO required review. The gate is the whole
345
+ # conscience in auto mode — the floor (min_delta), the suite
346
+ # no-regression phase, and the panel's taste rubric all bind
347
+ # BEFORE publish.
348
+ merge: Literal["manual", "auto"] = "manual"
349
+
350
+ @field_validator("benchmarks")
351
+ @classmethod
352
+ def _unique_names(cls, benchmarks: list[Benchmark]) -> list[Benchmark]:
353
+ # Benchmark names are IDENTITIES: they key branch names, ledger rows,
354
+ # and dispatched measure/job names. A duplicate silently collides all
355
+ # three (two measures share an eval dir; one result is lost).
356
+ names = [b.name for b in benchmarks]
357
+ dupes = sorted({n for n in names if names.count(n) > 1})
358
+ if dupes:
359
+ raise ValueError(f"duplicate benchmark name(s): {', '.join(dupes)}")
360
+ return benchmarks
361
+
362
+
363
+ _HOST_PREFIX = re.compile(r"^(?:[a-z+]+://)?(?:[^@/]*@)?(?:www\.)?github\.com[:/]+")
364
+
365
+
366
+ def normalize_repo(target_repo: str) -> str:
367
+ """Reduce a repo reference to `owner/name`, casefolded.
368
+
369
+ Accepts bare `owner/name`, trailing `/` or `.git`, and any scheme/userinfo
370
+ URL spelling, so the self-target check can't be dodged by spelling.
371
+ """
372
+ ref = target_repo.strip().casefold()
373
+ ref = _HOST_PREFIX.sub("", ref)
374
+ ref = re.sub(r"/{2,}", "/", ref).strip("/")
375
+ if ref.endswith(".git"):
376
+ ref = ref[: -len(".git")]
377
+ return ref
378
+
379
+
380
+ def normalize_path(entry: str) -> PurePosixPath:
381
+ """Normalize a scope entry, rejecting anything that can escape the repo."""
382
+ raw = entry.strip()
383
+ if not raw or raw in {".", "./"}:
384
+ raise ScopeError("allowing the repository root is never permitted")
385
+ if raw.startswith("/") or ":" in raw or "\\" in raw:
386
+ raise ScopeError(f"path must be repo-relative and POSIX: {entry!r}")
387
+ if _GLOB_CHARS & set(raw):
388
+ raise ScopeError(f"glob patterns are not allowed in scope: {entry!r}")
389
+ normalized = posixpath.normpath(raw)
390
+ path = PurePosixPath(normalized)
391
+ if normalized in {".", ""} or any(part == ".." for part in path.parts):
392
+ raise ScopeError(f"path escapes the repository: {entry!r}")
393
+ if any(part.casefold() == ".git" for part in path.parts):
394
+ raise ScopeError(f"the git directory is never writable: {entry!r}")
395
+ return path
396
+
397
+
398
+ def forbidden_paths(contract: Contract) -> tuple[str, ...]:
399
+ """Write paths forbidden for this contract: the hard-coded set plus the
400
+ target's roadmap."""
401
+ return (*ALWAYS_FORBIDDEN, str(normalize_path(contract.roadmap)))
402
+
403
+
404
+ def _fold(path: PurePosixPath) -> PurePosixPath:
405
+ """Casefold components: `.GITHUB/x` must be as forbidden as `.github/x`."""
406
+ return PurePosixPath(*[part.casefold() for part in path.parts])
407
+
408
+
409
+ def path_is_forbidden(candidate: str, contract: Contract) -> bool:
410
+ """True if `candidate` (a repo-relative file path) may not be written.
411
+
412
+ Component-wise and case-insensitive, so `README.mdx` is not shadowed by
413
+ roadmap `README.md` but `.GITHUB/` is still blocked. Anything
414
+ unnormalizable counts as forbidden.
415
+ """
416
+ try:
417
+ path = _fold(normalize_path(candidate))
418
+ except ScopeError:
419
+ return True
420
+ if any(part == ".git" for part in path.parts):
421
+ return True # never write into the git directory itself
422
+ for forbidden in forbidden_paths(contract):
423
+ f = _fold(PurePosixPath(forbidden))
424
+ if path == f or f in path.parents:
425
+ return True
426
+ return False
427
+
428
+
429
+ def _overlaps(allowed: PurePosixPath, forbidden: PurePosixPath) -> bool:
430
+ a, f = _fold(allowed), _fold(forbidden)
431
+ return a == f or f in a.parents or a in f.parents
432
+
433
+
434
+ def load_contract(text: str, target_repo: str) -> Contract:
435
+ """Parse and validate a contract for `target_repo`."""
436
+ if normalize_repo(target_repo) == SELF_REPO:
437
+ raise SelfTargetError("autoresearch is never a valid target of itself")
438
+ if len(text.encode()) > MAX_CONTRACT_BYTES:
439
+ raise ContractError(f"contract exceeds {MAX_CONTRACT_BYTES} bytes")
440
+ try:
441
+ data = yaml.load(text, Loader=_SafeLoader)
442
+ except ContractError:
443
+ raise
444
+ except (yaml.YAMLError, TypeError, ValueError) as exc:
445
+ raise ContractError(f"unparseable contract: {type(exc).__name__}") from None
446
+ if not isinstance(data, dict):
447
+ raise ContractError("contract must be a YAML mapping")
448
+ contract = Contract.model_validate(data)
449
+ forbidden = [PurePosixPath(p) for p in forbidden_paths(contract)]
450
+ for entry in contract.scope.allowed:
451
+ allowed = normalize_path(entry)
452
+ for path in forbidden:
453
+ if _overlaps(allowed, path):
454
+ raise ScopeError(f"allowed path {entry!r} overlaps forbidden {str(path)!r}")
455
+ # Shared paths route the suite gate off changed paths, which are already
456
+ # scope-checked — but a malformed entry would silently never match, so
457
+ # the same load-time rigor applies. A shared path the agent can never
458
+ # touch (no overlap with any allowed path) is dead config that reads as
459
+ # protection: refused loudly, like any contract typo.
460
+ solver_allowed = [normalize_path(entry) for entry in contract.scope.allowed]
461
+ for entry in contract.scope.shared:
462
+ shared = normalize_path(entry)
463
+ for path in forbidden:
464
+ if _overlaps(shared, path):
465
+ raise ScopeError(f"shared path {entry!r} overlaps forbidden {str(path)!r}")
466
+ if not any(_overlaps(shared, a) for a in solver_allowed):
467
+ raise ScopeError(
468
+ f"shared path {entry!r} overlaps no allowed path — the suite "
469
+ f"gate could never trigger on it"
470
+ )
471
+ if contract.steward is not None:
472
+ # same rigor as the solver scope, plus the role-separation invariant:
473
+ # steward and solver territories must not overlap AT LOAD TIME — a
474
+ # malformed or colliding entry is a contract error, never a
475
+ # mid-run surprise
476
+ solver = [normalize_path(entry) for entry in contract.scope.allowed]
477
+ records = [PurePosixPath(p) for p in RECORD_PATHS]
478
+ for entry in contract.steward.allowed:
479
+ allowed = normalize_path(entry)
480
+ for path in forbidden:
481
+ if _overlaps(allowed, path):
482
+ raise ScopeError(f"steward path {entry!r} overlaps forbidden {str(path)!r}")
483
+ for sp in solver:
484
+ if _overlaps(allowed, sp):
485
+ raise ScopeError(f"steward path {entry!r} overlaps solver scope {str(sp)!r}")
486
+ for rp in records:
487
+ if _overlaps(allowed, rp):
488
+ raise ScopeError(
489
+ f"steward path {entry!r} overlaps the record ledger "
490
+ f"{str(rp)!r} (orchestrator-owned)"
491
+ )
492
+ return contract
@@ -0,0 +1,63 @@
1
+ """Validate a contract file (`.outerloop.yaml`) before you push it.
2
+
3
+ uv run python -m outerloop.contract_cli .outerloop.yaml
4
+
5
+ Prints what the agent would be allowed to do, or exactly what is wrong.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import sys
11
+ from pathlib import Path
12
+
13
+ from outerloop.contract import (
14
+ CONTRACT_NAME,
15
+ CONTRACT_NAMES,
16
+ ContractError,
17
+ forbidden_paths,
18
+ load_contract,
19
+ )
20
+
21
+
22
+ def main(argv: list[str] | None = None) -> int:
23
+ args = argv if argv is not None else sys.argv[1:]
24
+ # no arg: whichever contract name the cwd has, new name first
25
+ path = (
26
+ Path(args[0])
27
+ if args
28
+ else next((Path(n) for n in CONTRACT_NAMES if Path(n).is_file()), Path(CONTRACT_NAME))
29
+ )
30
+ repo = args[1] if len(args) > 1 else "your-org/your-repo"
31
+
32
+ if not path.is_file():
33
+ print(f"✗ no contract at {path}")
34
+ return 2
35
+ try:
36
+ contract = load_contract(path.read_text(), repo)
37
+ except ContractError as exc:
38
+ print(f"✗ {path}: {exc}")
39
+ return 1
40
+
41
+ print(f"✓ {path} is valid\n")
42
+ print("Benchmarks the agent will try to improve:")
43
+ for benchmark in contract.benchmarks:
44
+ arrow = "↓ lower is better" if benchmark.direction == "min" else "↑ higher is better"
45
+ print(f" • {benchmark.name}: {benchmark.metric} ({arrow})")
46
+ print(f" $ {benchmark.command}")
47
+ if contract.suite is not None:
48
+ print(f"\nSuite aggregate: {contract.suite.metric} ({contract.suite.direction})")
49
+ print("\nThe agent may write only to:")
50
+ for allowed in contract.scope.allowed:
51
+ print(f" • {allowed}")
52
+ print("\nAlways forbidden, whatever the contract says:")
53
+ for forbidden in forbidden_paths(contract):
54
+ print(f" • {forbidden}")
55
+ print(
56
+ f"\nBudget: {contract.budgets.gpu_hours_per_run} GPU-hours per run, "
57
+ f"{contract.budgets.runs_per_week} runs per week"
58
+ )
59
+ return 0
60
+
61
+
62
+ if __name__ == "__main__":
63
+ sys.exit(main())