refractal 0.1.0a1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- refractal/__init__.py +109 -0
- refractal/build/__init__.py +479 -0
- refractal/cli.py +719 -0
- refractal/compare/__init__.py +66 -0
- refractal/compare/pairing.py +377 -0
- refractal/compare/stats.py +364 -0
- refractal/compare/variance.py +196 -0
- refractal/compare/verdict.py +589 -0
- refractal/example.py +98 -0
- refractal/execute/__init__.py +42 -0
- refractal/execute/fake.py +151 -0
- refractal/execute/harness.py +220 -0
- refractal/execute/local.py +163 -0
- refractal/execute/physics.py +126 -0
- refractal/execute/resources.py +122 -0
- refractal/execute/results.py +884 -0
- refractal/execute/vla_eval.py +940 -0
- refractal/execute/vla_eval_runner.py +694 -0
- refractal/expect.py +159 -0
- refractal/generators.py +24 -0
- refractal/init.py +236 -0
- refractal/perturbations/__init__.py +1056 -0
- refractal/predicates.py +34 -0
- refractal/promote.py +151 -0
- refractal/render/__init__.py +564 -0
- refractal/resolve/__init__.py +356 -0
- refractal/resolve/expand.py +432 -0
- refractal/resolve/fit.py +390 -0
- refractal/resolve/lock.py +314 -0
- refractal/schema/__init__.py +99 -0
- refractal/schema/canonical.py +212 -0
- refractal/schema/errors.py +57 -0
- refractal/schema/generators.py +258 -0
- refractal/schema/identity.py +428 -0
- refractal/schema/importstr.py +119 -0
- refractal/schema/loader.py +337 -0
- refractal/schema/models.py +767 -0
- refractal/schema/plan.py +349 -0
- refractal/sweep.py +180 -0
- refractal-0.1.0a1.dist-info/METADATA +314 -0
- refractal-0.1.0a1.dist-info/RECORD +45 -0
- refractal-0.1.0a1.dist-info/WHEEL +4 -0
- refractal-0.1.0a1.dist-info/entry_points.txt +2 -0
- refractal-0.1.0a1.dist-info/licenses/LICENSE +202 -0
- refractal-0.1.0a1.dist-info/licenses/NOTICE +23 -0
refractal/__init__.py
ADDED
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""Refractal -- scene-coherent placement and paired comparison for policy evaluation.
|
|
2
|
+
|
|
3
|
+
The harness answers *"what did this checkpoint score."* Refractal answers *"did
|
|
4
|
+
my change help."*
|
|
5
|
+
|
|
6
|
+
Refractal is a compiler, not a runtime: a catalog goes in, a ``plan.json`` comes
|
|
7
|
+
out, and the plan executes. The schedule is decided once, up front, and written
|
|
8
|
+
to a file you can read, diff and commit -- which for evaluation is better than a
|
|
9
|
+
runtime scheduler, because placement becomes part of the provenance.
|
|
10
|
+
|
|
11
|
+
Packages, ordered by what they are allowed to touch:
|
|
12
|
+
|
|
13
|
+
=============== ==================================== =====================
|
|
14
|
+
Package Does Touches infra?
|
|
15
|
+
=============== ==================================== =====================
|
|
16
|
+
``schema`` catalog, validation, identity no
|
|
17
|
+
``resolve`` catalog -> ``plan.json`` no
|
|
18
|
+
``execute`` runs a plan; backends yes, only this one
|
|
19
|
+
``compare`` Parquet -> verdict no
|
|
20
|
+
``cli`` wires them together --
|
|
21
|
+
=============== ==================================== =====================
|
|
22
|
+
|
|
23
|
+
Plan time versus render time
|
|
24
|
+
----------------------------
|
|
25
|
+
|
|
26
|
+
One rule decides which side of the ``plan.json`` boundary a fact belongs on:
|
|
27
|
+
|
|
28
|
+
**Anything that differs between two people running the same experiment is
|
|
29
|
+
render-time, not plan-time.**
|
|
30
|
+
|
|
31
|
+
A plan is a portable description of an experiment. If two people can execute the
|
|
32
|
+
same plan and get artifacts that are not interchangeable, the plan is carrying
|
|
33
|
+
something it should not.
|
|
34
|
+
|
|
35
|
+
Plan-time, and inside ``plan_id``: scenes, tasks, scenario sets, checkpoint ids,
|
|
36
|
+
seeds, seed base, tier. Everything that decides *which episodes exist*.
|
|
37
|
+
|
|
38
|
+
Render-time, and outside ``plan_id``: ``results_uri``, the uid and gid the
|
|
39
|
+
containers run as, ``execution_mode``, hardware profile, and every placement
|
|
40
|
+
decision derived from it -- device assignment, cpuset, worker count.
|
|
41
|
+
|
|
42
|
+
Placement is the edge case worth stating explicitly, because it *is* written
|
|
43
|
+
into ``plan.json`` and still must not enter ``plan_id``. The plan records it so
|
|
44
|
+
the schedule is diffable and reviewable before anything is spent, which is the
|
|
45
|
+
point of compiling rather than reconciling. But placement is hardware-dependent
|
|
46
|
+
and therefore not portable, so it is a rendering of the experiment rather than
|
|
47
|
+
part of its identity. Two people who plan the same catalog on different machines
|
|
48
|
+
get different worker layouts, the same ``plan_id``, and results that join.
|
|
49
|
+
|
|
50
|
+
Identity, precondition, provenance
|
|
51
|
+
----------------------------------
|
|
52
|
+
|
|
53
|
+
Three kinds of field, not two. The middle one is easy to collapse into either
|
|
54
|
+
neighbour and is where the interesting mistakes live.
|
|
55
|
+
|
|
56
|
+
============= ============= ================== ===================================
|
|
57
|
+
kind in ``plan_id`` gates? instances
|
|
58
|
+
============= ============= ================== ===================================
|
|
59
|
+
identity yes by construction ``scenario_hash``, ``task_hash``,
|
|
60
|
+
the checkpoint set, ``seeds``
|
|
61
|
+
precondition no yes, loudly ``scene_hash``, ``harness_version``
|
|
62
|
+
provenance no never ``catalog_hash``, ``measured_at``,
|
|
63
|
+
``source_sha``, ``session_id``
|
|
64
|
+
============= ============= ================== ===================================
|
|
65
|
+
|
|
66
|
+
**Identity** decides which episodes exist. Change one and you have a different
|
|
67
|
+
experiment, so results recorded before and after must not join -- and by
|
|
68
|
+
construction they cannot, because the ``plan_id`` differs.
|
|
69
|
+
|
|
70
|
+
**Precondition** decides whether a comparison is *meaningful*. It is kept out of
|
|
71
|
+
the key deliberately: putting ``scene_hash`` in the join key would make a mesh
|
|
72
|
+
edit produce an empty join, and an empty join is a legal result that raises
|
|
73
|
+
nothing. Kept out and checked separately, the same edit produces a sentence
|
|
74
|
+
somebody has to read. ``harness_version`` is the same shape -- vla-eval's own
|
|
75
|
+
paper reports a harness-side integration parameter moving a success rate by 55
|
|
76
|
+
points -- and gets the same treatment: out of ``plan_id``, because pinning it
|
|
77
|
+
there would invalidate every historical comparison on a dependency bump, but
|
|
78
|
+
gating ``compare``, because two runs from different harnesses may not be
|
|
79
|
+
comparable at all.
|
|
80
|
+
|
|
81
|
+
**Provenance** records where a fact came from. Its only job is to let a reader
|
|
82
|
+
tell an assertion from a measurement, and to let ``compare`` annotate.
|
|
83
|
+
|
|
84
|
+
Two questions separate them, and you need both:
|
|
85
|
+
|
|
86
|
+
1. *Would two runs differing only in this field be the same experiment?*
|
|
87
|
+
No -> identity.
|
|
88
|
+
2. *Should ``compare`` still produce a number?*
|
|
89
|
+
No -> precondition. Yes -> provenance.
|
|
90
|
+
|
|
91
|
+
``session_id`` answers yes to the second, with a note about warmup and thermal
|
|
92
|
+
state. ``scene_hash`` answers no. ``harness_version`` answers no, and that is a
|
|
93
|
+
decision taken explicitly rather than by omission -- with an override, because
|
|
94
|
+
most harness commits change no behaviour at all.
|
|
95
|
+
|
|
96
|
+
Both directions have been got wrong here already, which is why this is written
|
|
97
|
+
down: folding ``plan_schema`` into ``plan_id`` promoted provenance to identity,
|
|
98
|
+
and leaving a ``scene_hash`` mismatch as a note would have demoted a
|
|
99
|
+
precondition to provenance.
|
|
100
|
+
"""
|
|
101
|
+
|
|
102
|
+
try: # single source of truth is pyproject; this mirrors it at runtime
|
|
103
|
+
from importlib.metadata import PackageNotFoundError, version as _version
|
|
104
|
+
|
|
105
|
+
__version__ = _version("refractal")
|
|
106
|
+
except (ImportError, PackageNotFoundError): # running from a source tree
|
|
107
|
+
__version__ = "0.0.0+source"
|
|
108
|
+
|
|
109
|
+
__all__ = ["__version__"]
|
|
@@ -0,0 +1,479 @@
|
|
|
1
|
+
"""``refractal build`` -- the facts that need an environment `plan` does not have.
|
|
2
|
+
|
|
3
|
+
Four things drift, all for the same reason: they are properties of a machine
|
|
4
|
+
where the engine is installed, and `refractal plan` deliberately runs where it is
|
|
5
|
+
not.
|
|
6
|
+
|
|
7
|
+
===================== ======================================================
|
|
8
|
+
``scene_hash`` over model sources, meshes and the engine version
|
|
9
|
+
``engine_version`` only knowable where the engine exists
|
|
10
|
+
``resource_shape`` measured by running episodes
|
|
11
|
+
filter survivors may need FK, collision queries, penetration tests
|
|
12
|
+
===================== ======================================================
|
|
13
|
+
|
|
14
|
+
So one command produces all four, into ``catalog/build.lock``, and `resolve`
|
|
15
|
+
reads that and **refuses when an entry is stale** -- exactly as it refuses to
|
|
16
|
+
oversubscribe VRAM. A plan built on a stale fact is worse than no plan.
|
|
17
|
+
|
|
18
|
+
Two rules this establishes
|
|
19
|
+
--------------------------
|
|
20
|
+
|
|
21
|
+
**`build` and `execute` are the infra-touching packages; `schema`, `resolve` and
|
|
22
|
+
`compare` are not.** The original rule named only `execute`, which was written
|
|
23
|
+
before `build` existed. Both run where the engine does, and nothing else may.
|
|
24
|
+
|
|
25
|
+
**Hand-authored YAML is never rewritten.** The API reference has `build`
|
|
26
|
+
overwrite ``scenes.yaml`` in place. Machine-written and human-written data in one
|
|
27
|
+
file costs comments, produces a merge conflict on every rebuild, and makes it
|
|
28
|
+
impossible to tell what a human asserted from what a machine measured. Everything
|
|
29
|
+
derived goes in the lock.
|
|
30
|
+
|
|
31
|
+
What is *not* here yet
|
|
32
|
+
----------------------
|
|
33
|
+
|
|
34
|
+
Probing ``resource_shape`` means running episodes, which needs the adapter that
|
|
35
|
+
does not exist. The seam is defined (:class:`ShapeProber`) and the default
|
|
36
|
+
carries forward whatever ``scenes.yaml`` declares, recording that it was declared
|
|
37
|
+
rather than measured. `resolve` still refuses when no shape exists for the
|
|
38
|
+
requested hardware, so the failure stays loud.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
from __future__ import annotations
|
|
42
|
+
|
|
43
|
+
import datetime as dt
|
|
44
|
+
from dataclasses import dataclass, field
|
|
45
|
+
from pathlib import Path
|
|
46
|
+
from typing import Any, Protocol, Sequence
|
|
47
|
+
|
|
48
|
+
from ..resolve.lock import (
|
|
49
|
+
LOCK_FILENAME,
|
|
50
|
+
LOCK_SCHEMA,
|
|
51
|
+
BuildLock,
|
|
52
|
+
FilterEntry,
|
|
53
|
+
SceneEntry,
|
|
54
|
+
TaskEntry,
|
|
55
|
+
ShapeEntry,
|
|
56
|
+
filter_key,
|
|
57
|
+
filter_source_sha,
|
|
58
|
+
)
|
|
59
|
+
from ..schema.errors import CatalogError, RefractalError
|
|
60
|
+
from ..schema.canonical import hash_obj
|
|
61
|
+
from ..schema.identity import (
|
|
62
|
+
external_scene_ref_key,
|
|
63
|
+
scenario_hash,
|
|
64
|
+
scene_hash,
|
|
65
|
+
task_hash,
|
|
66
|
+
task_identity,
|
|
67
|
+
)
|
|
68
|
+
from ..schema.importstr import check_arity, resolve_import_string
|
|
69
|
+
from ..schema.loader import Catalog, load_catalog
|
|
70
|
+
from ..schema.models import ResourceShape, Scene
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
class BuildError(RefractalError):
|
|
74
|
+
"""Something needed for the lock could not be established."""
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class EngineProbe(Protocol):
|
|
78
|
+
"""Reports the engine version, and verifies a scene actually compiles.
|
|
79
|
+
|
|
80
|
+
The compile check is the other half of hashing sources rather than the
|
|
81
|
+
compiled model: `plan` hashes what it can read on a laptop, and whoever has
|
|
82
|
+
the engine confirms that those sources still produce the model they claim to.
|
|
83
|
+
"""
|
|
84
|
+
|
|
85
|
+
def version(self, engine: str) -> str: ...
|
|
86
|
+
|
|
87
|
+
def verify(self, engine: str, model_path: Path) -> None: ...
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
class ExternalSceneProbe(Protocol):
|
|
91
|
+
"""Supplies the facts that define a scene living inside a wrapped benchmark.
|
|
92
|
+
|
|
93
|
+
Returns a **document, not a digest**. Hashing is Refractal's job and happens
|
|
94
|
+
in exactly one place: a probe that hashed for itself would be a second
|
|
95
|
+
implementation of the canonicalisation, and two implementations of one rule
|
|
96
|
+
is how identities fork silently.
|
|
97
|
+
|
|
98
|
+
It also means the facts can be recorded in the lock alongside the hash, so a
|
|
99
|
+
changed ``scene_hash`` can be explained rather than merely observed -- the
|
|
100
|
+
same reason the harness surface records a manifest and not only a digest.
|
|
101
|
+
|
|
102
|
+
And it leaves a probe with nothing to import from Refractal. A probe is a
|
|
103
|
+
plugin *into* Refractal, unlike an adapter, which vla-eval loads and which
|
|
104
|
+
genuinely never touches it -- but nothing here needs it to depend on us.
|
|
105
|
+
"""
|
|
106
|
+
|
|
107
|
+
def external_scene_facts(self, scene: Scene) -> dict[str, Any]: ...
|
|
108
|
+
|
|
109
|
+
#: Optional. What can be done to the scene, for plan-time refusal of
|
|
110
|
+
#: perturbations. A probe without it simply declares nothing, and a plan
|
|
111
|
+
#: that perturbs such a scene is refused for want of a declaration rather
|
|
112
|
+
#: than allowed on the assumption that it would work.
|
|
113
|
+
def scene_capabilities(self, scene: Scene) -> dict[str, Any]: ...
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
class ShapeProber(Protocol):
|
|
117
|
+
"""Measures how much machine one worker of a scene needs."""
|
|
118
|
+
|
|
119
|
+
def measure(self, scene: Scene, hardware_profile: str) -> ResourceShape: ...
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _capabilities(probe: Any, scene: Scene, report: Any) -> dict[str, Any]:
|
|
123
|
+
"""Ask the probe what can be done to this scene, if it can say.
|
|
124
|
+
|
|
125
|
+
Absence is not an error here. It becomes one at plan time, and only for a
|
|
126
|
+
plan that actually perturbs the scene -- refusing a build because a probe
|
|
127
|
+
cannot describe actuators would break every catalog that never perturbs
|
|
128
|
+
anything, which is all of them today.
|
|
129
|
+
|
|
130
|
+
A probe that raises is reported and not fatal, for the same reason: reading
|
|
131
|
+
actuator limits means constructing the environment, which needs a GPU or an
|
|
132
|
+
EGL context that a build machine may not have.
|
|
133
|
+
"""
|
|
134
|
+
supplier = getattr(probe, "scene_capabilities", None)
|
|
135
|
+
if supplier is None:
|
|
136
|
+
return {}
|
|
137
|
+
try:
|
|
138
|
+
declared = supplier(scene)
|
|
139
|
+
except Exception as exc: # noqa: BLE001 - a probe failure must not fail the build
|
|
140
|
+
report.warnings.append(
|
|
141
|
+
f"scene {scene.id!r}: the probe could not report capabilities ({exc}). "
|
|
142
|
+
"A plan that perturbs this scene will be refused for want of a "
|
|
143
|
+
"declaration; one that does not is unaffected."
|
|
144
|
+
)
|
|
145
|
+
return {}
|
|
146
|
+
if not isinstance(declared, dict):
|
|
147
|
+
raise BuildError(
|
|
148
|
+
f"the probe returned {type(declared).__name__} for scene {scene.id!r}; "
|
|
149
|
+
"scene_capabilities must return a dict."
|
|
150
|
+
)
|
|
151
|
+
return declared
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
@dataclass
|
|
155
|
+
class DeclaredProbe:
|
|
156
|
+
"""No engine available: use what the catalog asserts, and record that.
|
|
157
|
+
|
|
158
|
+
Not a stub. Declaring a version by hand is legitimate -- it is how a catalog
|
|
159
|
+
stays reproducible on a machine that only plans -- and the lock records the
|
|
160
|
+
provenance so nobody later mistakes an assertion for a measurement.
|
|
161
|
+
"""
|
|
162
|
+
|
|
163
|
+
measured: bool = False
|
|
164
|
+
|
|
165
|
+
def version(self, engine: str) -> str:
|
|
166
|
+
raise BuildError(
|
|
167
|
+
f"no engine_version for {engine!r}: declare it in scenes.yaml, or run "
|
|
168
|
+
"'refractal build' where the engine is installed."
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
def verify(self, engine: str, model_path: Path) -> None:
|
|
172
|
+
return None
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
@dataclass
|
|
176
|
+
class BuildReport:
|
|
177
|
+
lock: BuildLock
|
|
178
|
+
warnings: list[str] = field(default_factory=list)
|
|
179
|
+
notes: list[str] = field(default_factory=list)
|
|
180
|
+
|
|
181
|
+
def summary_lines(self) -> list[str]:
|
|
182
|
+
lines = [f" computed scene_hash for {len(self.lock.scenes)} scene(s)"]
|
|
183
|
+
for entry in self.lock.filters:
|
|
184
|
+
lines.append(
|
|
185
|
+
f" {entry.scenario_set_id}: {entry.generated} generated, "
|
|
186
|
+
f"{entry.dropped} dropped by filter, {len(entry.survivors)} kept"
|
|
187
|
+
)
|
|
188
|
+
if self.lock.shapes:
|
|
189
|
+
lines.append(f" recorded resource_shape for {len(self.lock.shapes)} scene(s)")
|
|
190
|
+
return lines
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def build(
|
|
194
|
+
catalog: Catalog | str | Path,
|
|
195
|
+
*,
|
|
196
|
+
hardware_profile: str | None = None,
|
|
197
|
+
probe: EngineProbe | None = None,
|
|
198
|
+
prober: ShapeProber | None = None,
|
|
199
|
+
built_at: str | None = None,
|
|
200
|
+
write: bool = True,
|
|
201
|
+
) -> BuildReport:
|
|
202
|
+
"""Establish the four facts and write ``catalog/build.lock``.
|
|
203
|
+
|
|
204
|
+
``built_at`` is a parameter rather than a clock read, for the same reason
|
|
205
|
+
``created_at`` is on the planner: a command whose output depends on the time
|
|
206
|
+
cannot be tested for determinism.
|
|
207
|
+
"""
|
|
208
|
+
if not isinstance(catalog, Catalog):
|
|
209
|
+
catalog = load_catalog(catalog)
|
|
210
|
+
probe = probe or DeclaredProbe()
|
|
211
|
+
report = BuildReport(lock=BuildLock(lock_schema=LOCK_SCHEMA, built_at=built_at))
|
|
212
|
+
|
|
213
|
+
# --- 1 & 2: engine version, then scene hash over sources ------------
|
|
214
|
+
resolved_scene_hashes: dict[str, str] = {}
|
|
215
|
+
for scene in catalog.scenes:
|
|
216
|
+
version = scene.engine_version
|
|
217
|
+
if version is None:
|
|
218
|
+
version = probe.version(scene.engine)
|
|
219
|
+
report.notes.append(f"probed engine_version for {scene.id!r}: {version}")
|
|
220
|
+
scene = scene.model_copy(update={"engine_version": version})
|
|
221
|
+
|
|
222
|
+
if scene.is_external:
|
|
223
|
+
# The geometry lives in a wrapped benchmark. Only a probe that can
|
|
224
|
+
# import the provider knows what it is; without one there is nothing
|
|
225
|
+
# honest to record.
|
|
226
|
+
if probe is None or not hasattr(probe, "external_scene_facts"):
|
|
227
|
+
raise BuildError(
|
|
228
|
+
f"scene {scene.id!r} is defined by {scene.external.provider!r}, so its "
|
|
229
|
+
"facts must come from a probe that can import that provider. Run "
|
|
230
|
+
"'refractal build' where the benchmark is installed."
|
|
231
|
+
)
|
|
232
|
+
facts = probe.external_scene_facts(scene)
|
|
233
|
+
if not isinstance(facts, dict) or not facts:
|
|
234
|
+
raise BuildError(
|
|
235
|
+
f"the probe returned {type(facts).__name__} for scene {scene.id!r}; "
|
|
236
|
+
"external_scene_facts must return a non-empty dict of the facts that "
|
|
237
|
+
"define the scene. Refractal hashes it — a probe that returns a digest "
|
|
238
|
+
"would be a second implementation of the canonicalisation."
|
|
239
|
+
)
|
|
240
|
+
# The probe's facts AND the catalog's own assertions. The facts alone
|
|
241
|
+
# were the first version, and they are not enough: they describe what
|
|
242
|
+
# LIBERO contains, not how this catalog asks for it. `external.params`
|
|
243
|
+
# carries the benchmark's constructor arguments -- which cameras are
|
|
244
|
+
# sent, whether proprioception is sent, which of two quaternion
|
|
245
|
+
# conventions the state uses -- and those decide what the policy
|
|
246
|
+
# observes.
|
|
247
|
+
#
|
|
248
|
+
# Measured: `quat_no_antipodal` moves pi0 on one LIBERO task from 0/8
|
|
249
|
+
# to 2/4. Under the facts-only digest, a run with it and a run without
|
|
250
|
+
# it had the SAME scene_hash and the same plan_id, so they would have
|
|
251
|
+
# joined into one comparison and been averaged. That is the exact
|
|
252
|
+
# failure this project exists to make unrepresentable, sitting inside
|
|
253
|
+
# the identity scheme itself.
|
|
254
|
+
#
|
|
255
|
+
# Caught by checking plan_id after adding the flag rather than by the
|
|
256
|
+
# test, which asserted on `external_scene_ref_key` -- the helper --
|
|
257
|
+
# while the build hashed something else.
|
|
258
|
+
digest = hash_obj(
|
|
259
|
+
{"facts": facts, "catalog_ref": external_scene_ref_key(scene)}
|
|
260
|
+
)
|
|
261
|
+
resolved_scene_hashes[scene.id] = digest
|
|
262
|
+
report.lock.scenes.append(
|
|
263
|
+
SceneEntry(
|
|
264
|
+
scene_id=scene.id,
|
|
265
|
+
scene_hash=digest,
|
|
266
|
+
model_hash=digest,
|
|
267
|
+
engine_version=version,
|
|
268
|
+
external=True,
|
|
269
|
+
ref_key=external_scene_ref_key(scene),
|
|
270
|
+
facts=facts,
|
|
271
|
+
capabilities=_capabilities(probe, scene, report),
|
|
272
|
+
)
|
|
273
|
+
)
|
|
274
|
+
continue
|
|
275
|
+
|
|
276
|
+
digest = scene_hash(catalog.root, scene)
|
|
277
|
+
resolved_scene_hashes[scene.id] = digest
|
|
278
|
+
|
|
279
|
+
if scene.model_hash and scene.model_hash != digest:
|
|
280
|
+
# Loud: the sources moved since somebody wrote that value down, so
|
|
281
|
+
# every result recorded under it describes a different world.
|
|
282
|
+
report.warnings.append(
|
|
283
|
+
f"scene {scene.id!r} declares model_hash {scene.model_hash[:19]}... but its "
|
|
284
|
+
f"sources now hash to {digest[:19]}.... The geometry changed; results recorded "
|
|
285
|
+
"under the old hash are not comparable with new ones."
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
probe.verify(scene.engine, catalog.root / scene.model)
|
|
289
|
+
report.lock.scenes.append(
|
|
290
|
+
SceneEntry(
|
|
291
|
+
scene_id=scene.id,
|
|
292
|
+
scene_hash=digest,
|
|
293
|
+
model_hash=digest,
|
|
294
|
+
engine_version=version,
|
|
295
|
+
)
|
|
296
|
+
)
|
|
297
|
+
|
|
298
|
+
# --- 2b: task content, for goals the Task table cannot express ------
|
|
299
|
+
#
|
|
300
|
+
# `task_hash` covers instruction, predicate, arguments, step limit and
|
|
301
|
+
# phases -- complete for a task those fields DEFINE, and empty for one using
|
|
302
|
+
# `from_benchmark`, where the benchmark owns the definition. For a LIBERO
|
|
303
|
+
# task that leaves `instruction` as the only discriminator: a string a
|
|
304
|
+
# release could keep while moving the goal region underneath it.
|
|
305
|
+
#
|
|
306
|
+
# So the probe is asked, the same way it is asked for an external scene's
|
|
307
|
+
# facts, and the answer is hashed into `task_hash`. Probes that do not
|
|
308
|
+
# implement the hook are unaffected and their tasks keep the identity they
|
|
309
|
+
# had -- an empty content mapping hashes identically to none.
|
|
310
|
+
for task in catalog.tasks:
|
|
311
|
+
scene = catalog.scene(task.scene)
|
|
312
|
+
if not scene.is_external or probe is None:
|
|
313
|
+
continue
|
|
314
|
+
supplier = getattr(probe, "task_facts", None)
|
|
315
|
+
if supplier is None:
|
|
316
|
+
report.warnings.append(
|
|
317
|
+
f"task {task.id!r} runs on an externally-defined scene and its predicate "
|
|
318
|
+
f"is {task.predicate!r}, but the probe supplies no task_facts. Its "
|
|
319
|
+
"task_hash covers only the authored fields, so a provider release that "
|
|
320
|
+
"moved this goal without editing the instruction would not move any "
|
|
321
|
+
"identity."
|
|
322
|
+
)
|
|
323
|
+
continue
|
|
324
|
+
facts = supplier(scene, task)
|
|
325
|
+
if not isinstance(facts, dict):
|
|
326
|
+
raise BuildError(
|
|
327
|
+
f"the probe returned {type(facts).__name__} for task {task.id!r}; "
|
|
328
|
+
"task_facts must return a dict of the facts that define the goal. "
|
|
329
|
+
"Refractal hashes it -- a probe that returns a digest would be a second "
|
|
330
|
+
"implementation of the canonicalisation."
|
|
331
|
+
)
|
|
332
|
+
if not facts:
|
|
333
|
+
continue
|
|
334
|
+
report.lock.tasks.append(
|
|
335
|
+
TaskEntry(
|
|
336
|
+
task_id=task.id,
|
|
337
|
+
task_hash=task_hash(task, facts),
|
|
338
|
+
facts=facts,
|
|
339
|
+
authored_key=hash_obj(task_identity(task)),
|
|
340
|
+
)
|
|
341
|
+
)
|
|
342
|
+
if report.lock.tasks:
|
|
343
|
+
report.notes.append(
|
|
344
|
+
f"recorded provider facts for {len(report.lock.tasks)} task(s)"
|
|
345
|
+
)
|
|
346
|
+
|
|
347
|
+
# --- 3: filters, evaluated here because they may need the engine ----
|
|
348
|
+
for scenario_set in catalog.scenario_sets:
|
|
349
|
+
if scenario_set.filter is None:
|
|
350
|
+
continue
|
|
351
|
+
generator = resolve_import_string(scenario_set.generator)
|
|
352
|
+
candidates = generator(scenario_set.params, scenario_set.generator_seed)
|
|
353
|
+
try:
|
|
354
|
+
predicate = resolve_import_string(scenario_set.filter)
|
|
355
|
+
except Exception as exc:
|
|
356
|
+
raise BuildError(
|
|
357
|
+
f"scenario_set {scenario_set.id!r} declares filter {scenario_set.filter!r}, "
|
|
358
|
+
f"which cannot be imported here: {exc}. 'refractal build' must run where the "
|
|
359
|
+
"filter's dependencies are installed."
|
|
360
|
+
) from exc
|
|
361
|
+
|
|
362
|
+
# Arity before anything that runs the filter. Shape is a property of the
|
|
363
|
+
# callable alone; rejecting everything is a property of the callable and
|
|
364
|
+
# the grid together. Diagnose the simpler thing first, or a wrong-shaped
|
|
365
|
+
# filter reports as a workspace problem. Do not reorder these.
|
|
366
|
+
check_arity(predicate, "filter", scenario_set.filter)
|
|
367
|
+
|
|
368
|
+
survivors = []
|
|
369
|
+
for row in candidates:
|
|
370
|
+
try:
|
|
371
|
+
keep = bool(predicate(row))
|
|
372
|
+
except Exception as exc:
|
|
373
|
+
raise BuildError(
|
|
374
|
+
f"filter {scenario_set.filter!r} raised on scenario {row!r}: {exc}"
|
|
375
|
+
) from exc
|
|
376
|
+
if keep:
|
|
377
|
+
survivors.append(scenario_hash(row))
|
|
378
|
+
|
|
379
|
+
if not survivors:
|
|
380
|
+
# Naming the parameters is the difference between a useful error and
|
|
381
|
+
# a shrug. The third failure mode -- right arity, wrong parameter
|
|
382
|
+
# names -- is invisible otherwise: a filter reading `cube_x` from a
|
|
383
|
+
# grid that carries `vial_x` gets its default from `.get()` and
|
|
384
|
+
# rejects everything, which looks identical to an unreachable
|
|
385
|
+
# workspace. This was hit for real while writing the tests.
|
|
386
|
+
keys = sorted(candidates[0]) if candidates else []
|
|
387
|
+
raise BuildError(
|
|
388
|
+
f"filter {scenario_set.filter!r} rejected all {len(candidates)} scenarios in "
|
|
389
|
+
f"{scenario_set.id!r}. Those scenarios carry the parameters {keys}. "
|
|
390
|
+
"Either the filter reads parameters this grid does not define, the grid is "
|
|
391
|
+
"entirely outside the workspace, or the filter is inverted; all three are "
|
|
392
|
+
"worth knowing before a run, not after."
|
|
393
|
+
)
|
|
394
|
+
|
|
395
|
+
report.lock.filters.append(
|
|
396
|
+
FilterEntry(
|
|
397
|
+
scenario_set_id=scenario_set.id,
|
|
398
|
+
key=filter_key(scenario_set, resolved_scene_hashes[scenario_set.scene]),
|
|
399
|
+
survivors=survivors,
|
|
400
|
+
generated=len(candidates),
|
|
401
|
+
dropped=len(candidates) - len(survivors),
|
|
402
|
+
source_sha=filter_source_sha(predicate),
|
|
403
|
+
)
|
|
404
|
+
)
|
|
405
|
+
|
|
406
|
+
# --- 3b: predicates, checked where they are importable ---------------
|
|
407
|
+
# Best-effort: the adapter lives on the machine `build` runs on, but a
|
|
408
|
+
# catalog can legitimately be built before its adapter is installed. An
|
|
409
|
+
# unimportable predicate is a note; an importable one with the wrong shape
|
|
410
|
+
# is an error, because that is a real bug found for free.
|
|
411
|
+
for task in catalog.tasks:
|
|
412
|
+
for label, import_string in [("predicate", task.predicate)] + [
|
|
413
|
+
("predicate", phase.predicate) for phase in task.phases
|
|
414
|
+
]:
|
|
415
|
+
try:
|
|
416
|
+
fn = resolve_import_string(import_string)
|
|
417
|
+
except Exception:
|
|
418
|
+
report.notes.append(
|
|
419
|
+
f"could not import {label} {import_string!r} to check its shape"
|
|
420
|
+
)
|
|
421
|
+
continue
|
|
422
|
+
check_arity(fn, label, import_string)
|
|
423
|
+
|
|
424
|
+
# --- 4: resource shapes ---------------------------------------------
|
|
425
|
+
if hardware_profile:
|
|
426
|
+
for scene in catalog.scenes:
|
|
427
|
+
if prober is not None:
|
|
428
|
+
shape = prober.measure(scene, hardware_profile)
|
|
429
|
+
source = "measured"
|
|
430
|
+
else:
|
|
431
|
+
shape = scene.shape_for(hardware_profile)
|
|
432
|
+
source = "declared"
|
|
433
|
+
if shape is None:
|
|
434
|
+
report.warnings.append(
|
|
435
|
+
f"scene {scene.id!r} has no resource_shape for {hardware_profile!r} and "
|
|
436
|
+
"no prober was supplied, so none was recorded. 'refractal plan' will "
|
|
437
|
+
"refuse to fit workers for it."
|
|
438
|
+
)
|
|
439
|
+
continue
|
|
440
|
+
report.lock.shapes.append(
|
|
441
|
+
ShapeEntry(
|
|
442
|
+
scene_id=scene.id,
|
|
443
|
+
key=_shape_key(resolved_scene_hashes[scene.id], scene, hardware_profile),
|
|
444
|
+
shape=shape.model_copy(update={"measured_at": built_at if source == "measured" else None}),
|
|
445
|
+
)
|
|
446
|
+
)
|
|
447
|
+
report.notes.append(
|
|
448
|
+
f"resource_shape for {scene.id!r} on {hardware_profile!r}: {source}"
|
|
449
|
+
)
|
|
450
|
+
|
|
451
|
+
if write:
|
|
452
|
+
path = catalog.root / LOCK_FILENAME
|
|
453
|
+
path.write_text(report.lock.model_dump_json(indent=2) + "\n", encoding="utf-8")
|
|
454
|
+
|
|
455
|
+
return report
|
|
456
|
+
|
|
457
|
+
|
|
458
|
+
def _shape_key(scene_digest: str, scene: Scene, hardware_profile: str) -> str:
|
|
459
|
+
from ..schema.canonical import hash_obj
|
|
460
|
+
|
|
461
|
+
return hash_obj(
|
|
462
|
+
{
|
|
463
|
+
"scene_hash": scene_digest,
|
|
464
|
+
"engine": scene.engine,
|
|
465
|
+
"engine_version": scene.engine_version,
|
|
466
|
+
"hardware_profile": hardware_profile,
|
|
467
|
+
}
|
|
468
|
+
)
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
__all__ = [
|
|
472
|
+
"BuildError",
|
|
473
|
+
"BuildReport",
|
|
474
|
+
"DeclaredProbe",
|
|
475
|
+
"EngineProbe",
|
|
476
|
+
"ExternalSceneProbe",
|
|
477
|
+
"ShapeProber",
|
|
478
|
+
"build",
|
|
479
|
+
]
|