refractal 0.1.0a1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. refractal/__init__.py +109 -0
  2. refractal/build/__init__.py +479 -0
  3. refractal/cli.py +719 -0
  4. refractal/compare/__init__.py +66 -0
  5. refractal/compare/pairing.py +377 -0
  6. refractal/compare/stats.py +364 -0
  7. refractal/compare/variance.py +196 -0
  8. refractal/compare/verdict.py +589 -0
  9. refractal/example.py +98 -0
  10. refractal/execute/__init__.py +42 -0
  11. refractal/execute/fake.py +151 -0
  12. refractal/execute/harness.py +220 -0
  13. refractal/execute/local.py +163 -0
  14. refractal/execute/physics.py +126 -0
  15. refractal/execute/resources.py +122 -0
  16. refractal/execute/results.py +884 -0
  17. refractal/execute/vla_eval.py +940 -0
  18. refractal/execute/vla_eval_runner.py +694 -0
  19. refractal/expect.py +159 -0
  20. refractal/generators.py +24 -0
  21. refractal/init.py +236 -0
  22. refractal/perturbations/__init__.py +1056 -0
  23. refractal/predicates.py +34 -0
  24. refractal/promote.py +151 -0
  25. refractal/render/__init__.py +564 -0
  26. refractal/resolve/__init__.py +356 -0
  27. refractal/resolve/expand.py +432 -0
  28. refractal/resolve/fit.py +390 -0
  29. refractal/resolve/lock.py +314 -0
  30. refractal/schema/__init__.py +99 -0
  31. refractal/schema/canonical.py +212 -0
  32. refractal/schema/errors.py +57 -0
  33. refractal/schema/generators.py +258 -0
  34. refractal/schema/identity.py +428 -0
  35. refractal/schema/importstr.py +119 -0
  36. refractal/schema/loader.py +337 -0
  37. refractal/schema/models.py +767 -0
  38. refractal/schema/plan.py +349 -0
  39. refractal/sweep.py +180 -0
  40. refractal-0.1.0a1.dist-info/METADATA +314 -0
  41. refractal-0.1.0a1.dist-info/RECORD +45 -0
  42. refractal-0.1.0a1.dist-info/WHEEL +4 -0
  43. refractal-0.1.0a1.dist-info/entry_points.txt +2 -0
  44. refractal-0.1.0a1.dist-info/licenses/LICENSE +202 -0
  45. refractal-0.1.0a1.dist-info/licenses/NOTICE +23 -0
refractal/__init__.py ADDED
@@ -0,0 +1,109 @@
1
+ """Refractal -- scene-coherent placement and paired comparison for policy evaluation.
2
+
3
+ The harness answers *"what did this checkpoint score."* Refractal answers *"did
4
+ my change help."*
5
+
6
+ Refractal is a compiler, not a runtime: a catalog goes in, a ``plan.json`` comes
7
+ out, and the plan executes. The schedule is decided once, up front, and written
8
+ to a file you can read, diff and commit -- which for evaluation is better than a
9
+ runtime scheduler, because placement becomes part of the provenance.
10
+
11
+ Packages, ordered by what they are allowed to touch:
12
+
13
+ =============== ==================================== =====================
14
+ Package Does Touches infra?
15
+ =============== ==================================== =====================
16
+ ``schema`` catalog, validation, identity no
17
+ ``resolve`` catalog -> ``plan.json`` no
18
+ ``execute`` runs a plan; backends yes, only this one
19
+ ``compare`` Parquet -> verdict no
20
+ ``cli`` wires them together --
21
+ =============== ==================================== =====================
22
+
23
+ Plan time versus render time
24
+ ----------------------------
25
+
26
+ One rule decides which side of the ``plan.json`` boundary a fact belongs on:
27
+
28
+ **Anything that differs between two people running the same experiment is
29
+ render-time, not plan-time.**
30
+
31
+ A plan is a portable description of an experiment. If two people can execute the
32
+ same plan and get artifacts that are not interchangeable, the plan is carrying
33
+ something it should not.
34
+
35
+ Plan-time, and inside ``plan_id``: scenes, tasks, scenario sets, checkpoint ids,
36
+ seeds, seed base, tier. Everything that decides *which episodes exist*.
37
+
38
+ Render-time, and outside ``plan_id``: ``results_uri``, the uid and gid the
39
+ containers run as, ``execution_mode``, hardware profile, and every placement
40
+ decision derived from it -- device assignment, cpuset, worker count.
41
+
42
+ Placement is the edge case worth stating explicitly, because it *is* written
43
+ into ``plan.json`` and still must not enter ``plan_id``. The plan records it so
44
+ the schedule is diffable and reviewable before anything is spent, which is the
45
+ point of compiling rather than reconciling. But placement is hardware-dependent
46
+ and therefore not portable, so it is a rendering of the experiment rather than
47
+ part of its identity. Two people who plan the same catalog on different machines
48
+ get different worker layouts, the same ``plan_id``, and results that join.
49
+
50
+ Identity, precondition, provenance
51
+ ----------------------------------
52
+
53
+ Three kinds of field, not two. The middle one is easy to collapse into either
54
+ neighbour and is where the interesting mistakes live.
55
+
56
+ ============= ============= ================== ===================================
57
+ kind in ``plan_id`` gates? instances
58
+ ============= ============= ================== ===================================
59
+ identity yes by construction ``scenario_hash``, ``task_hash``,
60
+ the checkpoint set, ``seeds``
61
+ precondition no yes, loudly ``scene_hash``, ``harness_version``
62
+ provenance no never ``catalog_hash``, ``measured_at``,
63
+ ``source_sha``, ``session_id``
64
+ ============= ============= ================== ===================================
65
+
66
+ **Identity** decides which episodes exist. Change one and you have a different
67
+ experiment, so results recorded before and after must not join -- and by
68
+ construction they cannot, because the ``plan_id`` differs.
69
+
70
+ **Precondition** decides whether a comparison is *meaningful*. It is kept out of
71
+ the key deliberately: putting ``scene_hash`` in the join key would make a mesh
72
+ edit produce an empty join, and an empty join is a legal result that raises
73
+ nothing. Kept out and checked separately, the same edit produces a sentence
74
+ somebody has to read. ``harness_version`` is the same shape -- vla-eval's own
75
+ paper reports a harness-side integration parameter moving a success rate by 55
76
+ points -- and gets the same treatment: out of ``plan_id``, because pinning it
77
+ there would invalidate every historical comparison on a dependency bump, but
78
+ gating ``compare``, because two runs from different harnesses may not be
79
+ comparable at all.
80
+
81
+ **Provenance** records where a fact came from. Its only job is to let a reader
82
+ tell an assertion from a measurement, and to let ``compare`` annotate.
83
+
84
+ Two questions separate them, and you need both:
85
+
86
+ 1. *Would two runs differing only in this field be the same experiment?*
87
+ No -> identity.
88
+ 2. *Should ``compare`` still produce a number?*
89
+ No -> precondition. Yes -> provenance.
90
+
91
+ ``session_id`` answers yes to the second, with a note about warmup and thermal
92
+ state. ``scene_hash`` answers no. ``harness_version`` answers no, and that is a
93
+ decision taken explicitly rather than by omission -- with an override, because
94
+ most harness commits change no behaviour at all.
95
+
96
+ Both directions have been got wrong here already, which is why this is written
97
+ down: folding ``plan_schema`` into ``plan_id`` promoted provenance to identity,
98
+ and leaving a ``scene_hash`` mismatch as a note would have demoted a
99
+ precondition to provenance.
100
+ """
101
+
102
+ try: # single source of truth is pyproject; this mirrors it at runtime
103
+ from importlib.metadata import PackageNotFoundError, version as _version
104
+
105
+ __version__ = _version("refractal")
106
+ except (ImportError, PackageNotFoundError): # running from a source tree
107
+ __version__ = "0.0.0+source"
108
+
109
+ __all__ = ["__version__"]
@@ -0,0 +1,479 @@
1
+ """``refractal build`` -- the facts that need an environment `plan` does not have.
2
+
3
+ Four things drift, all for the same reason: they are properties of a machine
4
+ where the engine is installed, and `refractal plan` deliberately runs where it is
5
+ not.
6
+
7
+ ===================== ======================================================
8
+ ``scene_hash`` over model sources, meshes and the engine version
9
+ ``engine_version`` only knowable where the engine exists
10
+ ``resource_shape`` measured by running episodes
11
+ filter survivors may need FK, collision queries, penetration tests
12
+ ===================== ======================================================
13
+
14
+ So one command produces all four, into ``catalog/build.lock``, and `resolve`
15
+ reads that and **refuses when an entry is stale** -- exactly as it refuses to
16
+ oversubscribe VRAM. A plan built on a stale fact is worse than no plan.
17
+
18
+ Two rules this establishes
19
+ --------------------------
20
+
21
+ **`build` and `execute` are the infra-touching packages; `schema`, `resolve` and
22
+ `compare` are not.** The original rule named only `execute`, which was written
23
+ before `build` existed. Both run where the engine does, and nothing else may.
24
+
25
+ **Hand-authored YAML is never rewritten.** The API reference has `build`
26
+ overwrite ``scenes.yaml`` in place. Machine-written and human-written data in one
27
+ file costs comments, produces a merge conflict on every rebuild, and makes it
28
+ impossible to tell what a human asserted from what a machine measured. Everything
29
+ derived goes in the lock.
30
+
31
+ What is *not* here yet
32
+ ----------------------
33
+
34
+ Probing ``resource_shape`` means running episodes, which needs the adapter that
35
+ does not exist. The seam is defined (:class:`ShapeProber`) and the default
36
+ carries forward whatever ``scenes.yaml`` declares, recording that it was declared
37
+ rather than measured. `resolve` still refuses when no shape exists for the
38
+ requested hardware, so the failure stays loud.
39
+ """
40
+
41
+ from __future__ import annotations
42
+
43
+ import datetime as dt
44
+ from dataclasses import dataclass, field
45
+ from pathlib import Path
46
+ from typing import Any, Protocol, Sequence
47
+
48
+ from ..resolve.lock import (
49
+ LOCK_FILENAME,
50
+ LOCK_SCHEMA,
51
+ BuildLock,
52
+ FilterEntry,
53
+ SceneEntry,
54
+ TaskEntry,
55
+ ShapeEntry,
56
+ filter_key,
57
+ filter_source_sha,
58
+ )
59
+ from ..schema.errors import CatalogError, RefractalError
60
+ from ..schema.canonical import hash_obj
61
+ from ..schema.identity import (
62
+ external_scene_ref_key,
63
+ scenario_hash,
64
+ scene_hash,
65
+ task_hash,
66
+ task_identity,
67
+ )
68
+ from ..schema.importstr import check_arity, resolve_import_string
69
+ from ..schema.loader import Catalog, load_catalog
70
+ from ..schema.models import ResourceShape, Scene
71
+
72
+
73
+ class BuildError(RefractalError):
74
+ """Something needed for the lock could not be established."""
75
+
76
+
77
+ class EngineProbe(Protocol):
78
+ """Reports the engine version, and verifies a scene actually compiles.
79
+
80
+ The compile check is the other half of hashing sources rather than the
81
+ compiled model: `plan` hashes what it can read on a laptop, and whoever has
82
+ the engine confirms that those sources still produce the model they claim to.
83
+ """
84
+
85
+ def version(self, engine: str) -> str: ...
86
+
87
+ def verify(self, engine: str, model_path: Path) -> None: ...
88
+
89
+
90
+ class ExternalSceneProbe(Protocol):
91
+ """Supplies the facts that define a scene living inside a wrapped benchmark.
92
+
93
+ Returns a **document, not a digest**. Hashing is Refractal's job and happens
94
+ in exactly one place: a probe that hashed for itself would be a second
95
+ implementation of the canonicalisation, and two implementations of one rule
96
+ is how identities fork silently.
97
+
98
+ It also means the facts can be recorded in the lock alongside the hash, so a
99
+ changed ``scene_hash`` can be explained rather than merely observed -- the
100
+ same reason the harness surface records a manifest and not only a digest.
101
+
102
+ And it leaves a probe with nothing to import from Refractal. A probe is a
103
+ plugin *into* Refractal, unlike an adapter, which vla-eval loads and which
104
+ genuinely never touches it -- but nothing here needs it to depend on us.
105
+ """
106
+
107
+ def external_scene_facts(self, scene: Scene) -> dict[str, Any]: ...
108
+
109
+ #: Optional. What can be done to the scene, for plan-time refusal of
110
+ #: perturbations. A probe without it simply declares nothing, and a plan
111
+ #: that perturbs such a scene is refused for want of a declaration rather
112
+ #: than allowed on the assumption that it would work.
113
+ def scene_capabilities(self, scene: Scene) -> dict[str, Any]: ...
114
+
115
+
116
+ class ShapeProber(Protocol):
117
+ """Measures how much machine one worker of a scene needs."""
118
+
119
+ def measure(self, scene: Scene, hardware_profile: str) -> ResourceShape: ...
120
+
121
+
122
+ def _capabilities(probe: Any, scene: Scene, report: Any) -> dict[str, Any]:
123
+ """Ask the probe what can be done to this scene, if it can say.
124
+
125
+ Absence is not an error here. It becomes one at plan time, and only for a
126
+ plan that actually perturbs the scene -- refusing a build because a probe
127
+ cannot describe actuators would break every catalog that never perturbs
128
+ anything, which is all of them today.
129
+
130
+ A probe that raises is reported and not fatal, for the same reason: reading
131
+ actuator limits means constructing the environment, which needs a GPU or an
132
+ EGL context that a build machine may not have.
133
+ """
134
+ supplier = getattr(probe, "scene_capabilities", None)
135
+ if supplier is None:
136
+ return {}
137
+ try:
138
+ declared = supplier(scene)
139
+ except Exception as exc: # noqa: BLE001 - a probe failure must not fail the build
140
+ report.warnings.append(
141
+ f"scene {scene.id!r}: the probe could not report capabilities ({exc}). "
142
+ "A plan that perturbs this scene will be refused for want of a "
143
+ "declaration; one that does not is unaffected."
144
+ )
145
+ return {}
146
+ if not isinstance(declared, dict):
147
+ raise BuildError(
148
+ f"the probe returned {type(declared).__name__} for scene {scene.id!r}; "
149
+ "scene_capabilities must return a dict."
150
+ )
151
+ return declared
152
+
153
+
154
+ @dataclass
155
+ class DeclaredProbe:
156
+ """No engine available: use what the catalog asserts, and record that.
157
+
158
+ Not a stub. Declaring a version by hand is legitimate -- it is how a catalog
159
+ stays reproducible on a machine that only plans -- and the lock records the
160
+ provenance so nobody later mistakes an assertion for a measurement.
161
+ """
162
+
163
+ measured: bool = False
164
+
165
+ def version(self, engine: str) -> str:
166
+ raise BuildError(
167
+ f"no engine_version for {engine!r}: declare it in scenes.yaml, or run "
168
+ "'refractal build' where the engine is installed."
169
+ )
170
+
171
+ def verify(self, engine: str, model_path: Path) -> None:
172
+ return None
173
+
174
+
175
+ @dataclass
176
+ class BuildReport:
177
+ lock: BuildLock
178
+ warnings: list[str] = field(default_factory=list)
179
+ notes: list[str] = field(default_factory=list)
180
+
181
+ def summary_lines(self) -> list[str]:
182
+ lines = [f" computed scene_hash for {len(self.lock.scenes)} scene(s)"]
183
+ for entry in self.lock.filters:
184
+ lines.append(
185
+ f" {entry.scenario_set_id}: {entry.generated} generated, "
186
+ f"{entry.dropped} dropped by filter, {len(entry.survivors)} kept"
187
+ )
188
+ if self.lock.shapes:
189
+ lines.append(f" recorded resource_shape for {len(self.lock.shapes)} scene(s)")
190
+ return lines
191
+
192
+
193
+ def build(
194
+ catalog: Catalog | str | Path,
195
+ *,
196
+ hardware_profile: str | None = None,
197
+ probe: EngineProbe | None = None,
198
+ prober: ShapeProber | None = None,
199
+ built_at: str | None = None,
200
+ write: bool = True,
201
+ ) -> BuildReport:
202
+ """Establish the four facts and write ``catalog/build.lock``.
203
+
204
+ ``built_at`` is a parameter rather than a clock read, for the same reason
205
+ ``created_at`` is on the planner: a command whose output depends on the time
206
+ cannot be tested for determinism.
207
+ """
208
+ if not isinstance(catalog, Catalog):
209
+ catalog = load_catalog(catalog)
210
+ probe = probe or DeclaredProbe()
211
+ report = BuildReport(lock=BuildLock(lock_schema=LOCK_SCHEMA, built_at=built_at))
212
+
213
+ # --- 1 & 2: engine version, then scene hash over sources ------------
214
+ resolved_scene_hashes: dict[str, str] = {}
215
+ for scene in catalog.scenes:
216
+ version = scene.engine_version
217
+ if version is None:
218
+ version = probe.version(scene.engine)
219
+ report.notes.append(f"probed engine_version for {scene.id!r}: {version}")
220
+ scene = scene.model_copy(update={"engine_version": version})
221
+
222
+ if scene.is_external:
223
+ # The geometry lives in a wrapped benchmark. Only a probe that can
224
+ # import the provider knows what it is; without one there is nothing
225
+ # honest to record.
226
+ if probe is None or not hasattr(probe, "external_scene_facts"):
227
+ raise BuildError(
228
+ f"scene {scene.id!r} is defined by {scene.external.provider!r}, so its "
229
+ "facts must come from a probe that can import that provider. Run "
230
+ "'refractal build' where the benchmark is installed."
231
+ )
232
+ facts = probe.external_scene_facts(scene)
233
+ if not isinstance(facts, dict) or not facts:
234
+ raise BuildError(
235
+ f"the probe returned {type(facts).__name__} for scene {scene.id!r}; "
236
+ "external_scene_facts must return a non-empty dict of the facts that "
237
+ "define the scene. Refractal hashes it — a probe that returns a digest "
238
+ "would be a second implementation of the canonicalisation."
239
+ )
240
+ # The probe's facts AND the catalog's own assertions. The facts alone
241
+ # were the first version, and they are not enough: they describe what
242
+ # LIBERO contains, not how this catalog asks for it. `external.params`
243
+ # carries the benchmark's constructor arguments -- which cameras are
244
+ # sent, whether proprioception is sent, which of two quaternion
245
+ # conventions the state uses -- and those decide what the policy
246
+ # observes.
247
+ #
248
+ # Measured: `quat_no_antipodal` moves pi0 on one LIBERO task from 0/8
249
+ # to 2/4. Under the facts-only digest, a run with it and a run without
250
+ # it had the SAME scene_hash and the same plan_id, so they would have
251
+ # joined into one comparison and been averaged. That is the exact
252
+ # failure this project exists to make unrepresentable, sitting inside
253
+ # the identity scheme itself.
254
+ #
255
+ # Caught by checking plan_id after adding the flag rather than by the
256
+ # test, which asserted on `external_scene_ref_key` -- the helper --
257
+ # while the build hashed something else.
258
+ digest = hash_obj(
259
+ {"facts": facts, "catalog_ref": external_scene_ref_key(scene)}
260
+ )
261
+ resolved_scene_hashes[scene.id] = digest
262
+ report.lock.scenes.append(
263
+ SceneEntry(
264
+ scene_id=scene.id,
265
+ scene_hash=digest,
266
+ model_hash=digest,
267
+ engine_version=version,
268
+ external=True,
269
+ ref_key=external_scene_ref_key(scene),
270
+ facts=facts,
271
+ capabilities=_capabilities(probe, scene, report),
272
+ )
273
+ )
274
+ continue
275
+
276
+ digest = scene_hash(catalog.root, scene)
277
+ resolved_scene_hashes[scene.id] = digest
278
+
279
+ if scene.model_hash and scene.model_hash != digest:
280
+ # Loud: the sources moved since somebody wrote that value down, so
281
+ # every result recorded under it describes a different world.
282
+ report.warnings.append(
283
+ f"scene {scene.id!r} declares model_hash {scene.model_hash[:19]}... but its "
284
+ f"sources now hash to {digest[:19]}.... The geometry changed; results recorded "
285
+ "under the old hash are not comparable with new ones."
286
+ )
287
+
288
+ probe.verify(scene.engine, catalog.root / scene.model)
289
+ report.lock.scenes.append(
290
+ SceneEntry(
291
+ scene_id=scene.id,
292
+ scene_hash=digest,
293
+ model_hash=digest,
294
+ engine_version=version,
295
+ )
296
+ )
297
+
298
+ # --- 2b: task content, for goals the Task table cannot express ------
299
+ #
300
+ # `task_hash` covers instruction, predicate, arguments, step limit and
301
+ # phases -- complete for a task those fields DEFINE, and empty for one using
302
+ # `from_benchmark`, where the benchmark owns the definition. For a LIBERO
303
+ # task that leaves `instruction` as the only discriminator: a string a
304
+ # release could keep while moving the goal region underneath it.
305
+ #
306
+ # So the probe is asked, the same way it is asked for an external scene's
307
+ # facts, and the answer is hashed into `task_hash`. Probes that do not
308
+ # implement the hook are unaffected and their tasks keep the identity they
309
+ # had -- an empty content mapping hashes identically to none.
310
+ for task in catalog.tasks:
311
+ scene = catalog.scene(task.scene)
312
+ if not scene.is_external or probe is None:
313
+ continue
314
+ supplier = getattr(probe, "task_facts", None)
315
+ if supplier is None:
316
+ report.warnings.append(
317
+ f"task {task.id!r} runs on an externally-defined scene and its predicate "
318
+ f"is {task.predicate!r}, but the probe supplies no task_facts. Its "
319
+ "task_hash covers only the authored fields, so a provider release that "
320
+ "moved this goal without editing the instruction would not move any "
321
+ "identity."
322
+ )
323
+ continue
324
+ facts = supplier(scene, task)
325
+ if not isinstance(facts, dict):
326
+ raise BuildError(
327
+ f"the probe returned {type(facts).__name__} for task {task.id!r}; "
328
+ "task_facts must return a dict of the facts that define the goal. "
329
+ "Refractal hashes it -- a probe that returns a digest would be a second "
330
+ "implementation of the canonicalisation."
331
+ )
332
+ if not facts:
333
+ continue
334
+ report.lock.tasks.append(
335
+ TaskEntry(
336
+ task_id=task.id,
337
+ task_hash=task_hash(task, facts),
338
+ facts=facts,
339
+ authored_key=hash_obj(task_identity(task)),
340
+ )
341
+ )
342
+ if report.lock.tasks:
343
+ report.notes.append(
344
+ f"recorded provider facts for {len(report.lock.tasks)} task(s)"
345
+ )
346
+
347
+ # --- 3: filters, evaluated here because they may need the engine ----
348
+ for scenario_set in catalog.scenario_sets:
349
+ if scenario_set.filter is None:
350
+ continue
351
+ generator = resolve_import_string(scenario_set.generator)
352
+ candidates = generator(scenario_set.params, scenario_set.generator_seed)
353
+ try:
354
+ predicate = resolve_import_string(scenario_set.filter)
355
+ except Exception as exc:
356
+ raise BuildError(
357
+ f"scenario_set {scenario_set.id!r} declares filter {scenario_set.filter!r}, "
358
+ f"which cannot be imported here: {exc}. 'refractal build' must run where the "
359
+ "filter's dependencies are installed."
360
+ ) from exc
361
+
362
+ # Arity before anything that runs the filter. Shape is a property of the
363
+ # callable alone; rejecting everything is a property of the callable and
364
+ # the grid together. Diagnose the simpler thing first, or a wrong-shaped
365
+ # filter reports as a workspace problem. Do not reorder these.
366
+ check_arity(predicate, "filter", scenario_set.filter)
367
+
368
+ survivors = []
369
+ for row in candidates:
370
+ try:
371
+ keep = bool(predicate(row))
372
+ except Exception as exc:
373
+ raise BuildError(
374
+ f"filter {scenario_set.filter!r} raised on scenario {row!r}: {exc}"
375
+ ) from exc
376
+ if keep:
377
+ survivors.append(scenario_hash(row))
378
+
379
+ if not survivors:
380
+ # Naming the parameters is the difference between a useful error and
381
+ # a shrug. The third failure mode -- right arity, wrong parameter
382
+ # names -- is invisible otherwise: a filter reading `cube_x` from a
383
+ # grid that carries `vial_x` gets its default from `.get()` and
384
+ # rejects everything, which looks identical to an unreachable
385
+ # workspace. This was hit for real while writing the tests.
386
+ keys = sorted(candidates[0]) if candidates else []
387
+ raise BuildError(
388
+ f"filter {scenario_set.filter!r} rejected all {len(candidates)} scenarios in "
389
+ f"{scenario_set.id!r}. Those scenarios carry the parameters {keys}. "
390
+ "Either the filter reads parameters this grid does not define, the grid is "
391
+ "entirely outside the workspace, or the filter is inverted; all three are "
392
+ "worth knowing before a run, not after."
393
+ )
394
+
395
+ report.lock.filters.append(
396
+ FilterEntry(
397
+ scenario_set_id=scenario_set.id,
398
+ key=filter_key(scenario_set, resolved_scene_hashes[scenario_set.scene]),
399
+ survivors=survivors,
400
+ generated=len(candidates),
401
+ dropped=len(candidates) - len(survivors),
402
+ source_sha=filter_source_sha(predicate),
403
+ )
404
+ )
405
+
406
+ # --- 3b: predicates, checked where they are importable ---------------
407
+ # Best-effort: the adapter lives on the machine `build` runs on, but a
408
+ # catalog can legitimately be built before its adapter is installed. An
409
+ # unimportable predicate is a note; an importable one with the wrong shape
410
+ # is an error, because that is a real bug found for free.
411
+ for task in catalog.tasks:
412
+ for label, import_string in [("predicate", task.predicate)] + [
413
+ ("predicate", phase.predicate) for phase in task.phases
414
+ ]:
415
+ try:
416
+ fn = resolve_import_string(import_string)
417
+ except Exception:
418
+ report.notes.append(
419
+ f"could not import {label} {import_string!r} to check its shape"
420
+ )
421
+ continue
422
+ check_arity(fn, label, import_string)
423
+
424
+ # --- 4: resource shapes ---------------------------------------------
425
+ if hardware_profile:
426
+ for scene in catalog.scenes:
427
+ if prober is not None:
428
+ shape = prober.measure(scene, hardware_profile)
429
+ source = "measured"
430
+ else:
431
+ shape = scene.shape_for(hardware_profile)
432
+ source = "declared"
433
+ if shape is None:
434
+ report.warnings.append(
435
+ f"scene {scene.id!r} has no resource_shape for {hardware_profile!r} and "
436
+ "no prober was supplied, so none was recorded. 'refractal plan' will "
437
+ "refuse to fit workers for it."
438
+ )
439
+ continue
440
+ report.lock.shapes.append(
441
+ ShapeEntry(
442
+ scene_id=scene.id,
443
+ key=_shape_key(resolved_scene_hashes[scene.id], scene, hardware_profile),
444
+ shape=shape.model_copy(update={"measured_at": built_at if source == "measured" else None}),
445
+ )
446
+ )
447
+ report.notes.append(
448
+ f"resource_shape for {scene.id!r} on {hardware_profile!r}: {source}"
449
+ )
450
+
451
+ if write:
452
+ path = catalog.root / LOCK_FILENAME
453
+ path.write_text(report.lock.model_dump_json(indent=2) + "\n", encoding="utf-8")
454
+
455
+ return report
456
+
457
+
458
+ def _shape_key(scene_digest: str, scene: Scene, hardware_profile: str) -> str:
459
+ from ..schema.canonical import hash_obj
460
+
461
+ return hash_obj(
462
+ {
463
+ "scene_hash": scene_digest,
464
+ "engine": scene.engine,
465
+ "engine_version": scene.engine_version,
466
+ "hardware_profile": hardware_profile,
467
+ }
468
+ )
469
+
470
+
471
+ __all__ = [
472
+ "BuildError",
473
+ "BuildReport",
474
+ "DeclaredProbe",
475
+ "EngineProbe",
476
+ "ExternalSceneProbe",
477
+ "ShapeProber",
478
+ "build",
479
+ ]