techtree 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- techtree/__init__.py +35 -0
- techtree/__main__.py +14 -0
- techtree/canonical.py +239 -0
- techtree/catalog/__init__.py +25 -0
- techtree/catalog/repository.py +400 -0
- techtree/catalog/service.py +419 -0
- techtree/cli/__init__.py +1 -0
- techtree/cli/app.py +416 -0
- techtree/cli/commands/__init__.py +1 -0
- techtree/cli/commands/climb.py +1223 -0
- techtree/cli/commands/doctor.py +147 -0
- techtree/cli/commands/engine.py +207 -0
- techtree/cli/commands/proof.py +556 -0
- techtree/cli/commands/publish.py +447 -0
- techtree/cli/commands/release.py +303 -0
- techtree/cli/commands/run.py +1067 -0
- techtree/cli/commands/setup.py +181 -0
- techtree/cli/commands/skill.py +221 -0
- techtree/cli/commands/uplift.py +698 -0
- techtree/cli/commands/withdraw.py +212 -0
- techtree/cli/confirm.py +47 -0
- techtree/cli/context.py +96 -0
- techtree/cli/invoke.py +220 -0
- techtree/cli/output.py +280 -0
- techtree/constants.py +138 -0
- techtree/crypto.py +128 -0
- techtree/doctor/__init__.py +1 -0
- techtree/doctor/checks.py +675 -0
- techtree/doctor/execution_checks.py +435 -0
- techtree/doctor/service.py +326 -0
- techtree/drafts/__init__.py +32 -0
- techtree/drafts/source.py +146 -0
- techtree/drafts/store.py +992 -0
- techtree/engines/__init__.py +1 -0
- techtree/engines/bundle.py +251 -0
- techtree/engines/installer.py +679 -0
- techtree/engines/registry.py +235 -0
- techtree/engines/runner.py +170 -0
- techtree/errors.py +262 -0
- techtree/fs.py +234 -0
- techtree/harness.py +108 -0
- techtree/identity/__init__.py +41 -0
- techtree/identity/models.py +113 -0
- techtree/identity/service.py +199 -0
- techtree/identity/store.py +263 -0
- techtree/ids.py +85 -0
- techtree/manifests/__init__.py +39 -0
- techtree/manifests/builder.py +433 -0
- techtree/manifests/compare.py +376 -0
- techtree/models/__init__.py +282 -0
- techtree/models/base.py +201 -0
- techtree/models/campaign.py +484 -0
- techtree/models/catalog.py +227 -0
- techtree/models/cli.py +151 -0
- techtree/models/climb.py +254 -0
- techtree/models/data_policy.py +130 -0
- techtree/models/engine.py +156 -0
- techtree/models/episode_receipt.py +130 -0
- techtree/models/evaluation_backend.py +113 -0
- techtree/models/experiment.py +154 -0
- techtree/models/run.py +214 -0
- techtree/models/skill.py +156 -0
- techtree/models/uplift_report.py +158 -0
- techtree/models/validation.py +299 -0
- techtree/paths.py +116 -0
- techtree/presentation/__init__.py +31 -0
- techtree/presentation/build.py +1242 -0
- techtree/presentation/compact.py +246 -0
- techtree/presentation/evidence.py +169 -0
- techtree/presentation/models.py +358 -0
- techtree/presentation/rich.py +312 -0
- techtree/presentation/sanitize.py +156 -0
- techtree/publication/__init__.py +44 -0
- techtree/publication/address.py +180 -0
- techtree/publication/coordinates.py +26 -0
- techtree/publication/journal.py +212 -0
- techtree/publication/keccak.py +183 -0
- techtree/publication/models.py +209 -0
- techtree/publication/offer.py +35 -0
- techtree/publication/service.py +618 -0
- techtree/publication/transport.py +296 -0
- techtree/publication/verify.py +242 -0
- techtree/publication/withdraw.py +156 -0
- techtree/py.typed +0 -0
- techtree/receipts/__init__.py +52 -0
- techtree/receipts/bundle.py +578 -0
- techtree/receipts/compare.py +1065 -0
- techtree/receipts/episode.py +672 -0
- techtree/receipts/execution.py +630 -0
- techtree/receipts/observed.py +474 -0
- techtree/receipts/set.py +336 -0
- techtree/receipts/uplift.py +655 -0
- techtree/receipts/verify.py +1055 -0
- techtree/release/__init__.py +9 -0
- techtree/release/bootstrap.py +509 -0
- techtree/release/checks.py +376 -0
- techtree/release/document.py +125 -0
- techtree/release/generate.py +221 -0
- techtree/release/models.py +293 -0
- techtree/release/provenance.py +109 -0
- techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
- techtree/resources/catalog/catalog.json +32 -0
- techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
- techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
- techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
- techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
- techtree/resources/engines/default/engine.json +20 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
- techtree/resources/engines/default/pyproject.toml +23 -0
- techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
- techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
- techtree/resources/engines/default/tools/normalize_validation.py +222 -0
- techtree/resources/engines/default/uv.lock +1758 -0
- techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
- techtree/resources/release/build-provenance.json +4 -0
- techtree/resources/release/release-core.json +24 -0
- techtree/runs/__init__.py +31 -0
- techtree/runs/artifacts.py +750 -0
- techtree/runs/child_registry.py +228 -0
- techtree/runs/events.py +478 -0
- techtree/runs/executor.py +140 -0
- techtree/runs/fake.py +741 -0
- techtree/runs/launcher.py +253 -0
- techtree/runs/machine.py +489 -0
- techtree/runs/real.py +789 -0
- techtree/runs/service.py +616 -0
- techtree/runs/store.py +555 -0
- techtree/runs/validation.py +259 -0
- techtree/runs/variants.py +684 -0
- techtree/settings.py +143 -0
- techtree/skills/__init__.py +14 -0
- techtree/skills/archive.py +282 -0
- techtree/skills/policy.py +62 -0
- techtree/skills/scanner.py +394 -0
- techtree/skills/service.py +752 -0
- techtree/skills/starter.py +434 -0
- techtree/tasksets/__init__.py +1 -0
- techtree/tasksets/membership.py +269 -0
- techtree/tasksets/provider.py +207 -0
- techtree/tasksets/resolver.py +311 -0
- techtree/tasksets/service.py +484 -0
- techtree/tasksets/verifiers_cli.py +538 -0
- techtree/uplift/__init__.py +20 -0
- techtree/uplift/context.py +544 -0
- techtree/uplift/derive.py +203 -0
- techtree/uplift/public_tasks.py +151 -0
- techtree/uplift/service.py +719 -0
- techtree/uplift/source.py +160 -0
- techtree/verifiers/__init__.py +31 -0
- techtree/verifiers/budget.py +219 -0
- techtree/verifiers/child.py +633 -0
- techtree/verifiers/compiler.py +432 -0
- techtree/verifiers/config.py +365 -0
- techtree/verifiers/credentials.py +321 -0
- techtree/verifiers/image.py +126 -0
- techtree/verifiers/models.py +527 -0
- techtree/verifiers/outputs.py +368 -0
- techtree/verifiers/progress.py +192 -0
- techtree/verifiers/supervisor.py +341 -0
- techtree/verifiers/verify.py +782 -0
- techtree/version.py +39 -0
- techtree/worker/__init__.py +18 -0
- techtree/worker/execute.py +487 -0
- techtree/worker/main.py +57 -0
- techtree-0.1.0.dist-info/METADATA +344 -0
- techtree-0.1.0.dist-info/RECORD +174 -0
- techtree-0.1.0.dist-info/WHEEL +4 -0
- techtree-0.1.0.dist-info/entry_points.txt +3 -0
- techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,782 @@
|
|
|
1
|
+
"""Whether a compiled configuration survives the engine. Spec section 6.14.
|
|
2
|
+
|
|
3
|
+
The dry run is the only cheap way to ask the pinned engine what it thinks of a
|
|
4
|
+
Techtree configuration: it resolves the taskset plugin, narrows the environment
|
|
5
|
+
config to the real ``Env`` class, rejects any key the model does not declare,
|
|
6
|
+
and writes back the configuration it would actually run. It costs nothing, it
|
|
7
|
+
touches no provider, and it is where a mistyped taskset id or a seat the
|
|
8
|
+
environment does not declare surfaces in a second rather than after Docker has
|
|
9
|
+
been provisioned.
|
|
10
|
+
|
|
11
|
+
It is not, however, a validation of the *experiment*. Four things Techtree
|
|
12
|
+
cares about pass a dry run cleanly and fail later or never: a bundled skill
|
|
13
|
+
catalogue, ``disabled_tools``, a skill path that does not exist, and a config
|
|
14
|
+
with no taskset at all (``docs/verifiers-eval.md``, finding E3). Those are
|
|
15
|
+
rejected by :mod:`techtree.verifiers.config` and
|
|
16
|
+
:mod:`techtree.verifiers.compiler`, before anything is written. This module
|
|
17
|
+
asks the complementary question, and the two together are what section 6.14's
|
|
18
|
+
pre-execution half needs.
|
|
19
|
+
|
|
20
|
+
The resolved configuration is compared to the compiled one as a **projection**,
|
|
21
|
+
never byte for byte. The engine fills in ``client.base_url`` that Techtree
|
|
22
|
+
deliberately omitted, may add a routing header of its own, and writes out every
|
|
23
|
+
default Techtree never mentioned. It also records the ``--output-dir`` given on
|
|
24
|
+
argv rather than the one in the file, so the dry run's own redirection is
|
|
25
|
+
folded into the comparison instead of being excused from it. What must hold is
|
|
26
|
+
that nothing Techtree *did* declare came back changed.
|
|
27
|
+
|
|
28
|
+
Two settings are checked in the resolved document even though the compiled one
|
|
29
|
+
never mentions them, because for both the engine's default is the dangerous
|
|
30
|
+
answer and a flag is what turns it off: the platform upload, and whether the
|
|
31
|
+
rollouts are hosted through an env-server worker pool. The dry run carries the
|
|
32
|
+
same flags the run does, so what it resolves is what the run would do — and a
|
|
33
|
+
flag the engine stopped understanding is found here, before anything is spent,
|
|
34
|
+
rather than afterwards.
|
|
35
|
+
|
|
36
|
+
Section 6.14's post-execution half is :func:`verify_variant_execution`. It
|
|
37
|
+
answers one question — is this execution complete and scientifically usable —
|
|
38
|
+
as an ordered list of named verdicts rather than as an exception, so a caller
|
|
39
|
+
reads which rule failed instead of catching something and guessing. It raises
|
|
40
|
+
only on inputs too malformed to check at all.
|
|
41
|
+
|
|
42
|
+
A zero exit code is never sufficient. The output checks decide validity, and a
|
|
43
|
+
graceful cancellation maps to cancellation rather than to a scientific verdict:
|
|
44
|
+
a run somebody stopped did not produce a wrong answer, it produced no answer.
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
from __future__ import annotations
|
|
48
|
+
|
|
49
|
+
import json
|
|
50
|
+
from collections.abc import Mapping
|
|
51
|
+
from dataclasses import dataclass
|
|
52
|
+
from pathlib import Path
|
|
53
|
+
from typing import Any, Final
|
|
54
|
+
|
|
55
|
+
from techtree.engines.runner import EngineProcessResult, EngineRunner
|
|
56
|
+
from techtree.errors import ValidationError
|
|
57
|
+
from techtree.fs import ensure_private_directory
|
|
58
|
+
from techtree.models.campaign import SUBJECT_AGENT, AgentSpec, ModelSpec
|
|
59
|
+
from techtree.models.engine import EngineDescriptor
|
|
60
|
+
from techtree.models.experiment import ExperimentManifest
|
|
61
|
+
from techtree.models.validation import TasksetLock
|
|
62
|
+
from techtree.verifiers.child import (
|
|
63
|
+
CANCELLATION_EXIT_CODE,
|
|
64
|
+
DRY_RUN_NAME,
|
|
65
|
+
EVAL_EXECUTABLE,
|
|
66
|
+
dry_run_argv,
|
|
67
|
+
write_command_log,
|
|
68
|
+
)
|
|
69
|
+
from techtree.verifiers.config import EvalToml, emitted_document
|
|
70
|
+
from techtree.verifiers.credentials import credential_status
|
|
71
|
+
from techtree.verifiers.models import (
|
|
72
|
+
COMMAND_LOG_FILENAME,
|
|
73
|
+
VERIFIERS_DIRECTORY,
|
|
74
|
+
ExecutionCheck,
|
|
75
|
+
NormalizedTrace,
|
|
76
|
+
VariantExecutionResult,
|
|
77
|
+
VariantName,
|
|
78
|
+
)
|
|
79
|
+
from techtree.verifiers.outputs import RESOLVED_CONFIG_PATH
|
|
80
|
+
|
|
81
|
+
__all__ = [
|
|
82
|
+
"DEFAULT_DRY_RUN_TIMEOUT_SECONDS",
|
|
83
|
+
"VARIANT_DRY_RUN_FAILED",
|
|
84
|
+
"VARIANT_EXECUTION_UNCHECKABLE",
|
|
85
|
+
"DryRunOutcome",
|
|
86
|
+
"dry_run_variant_config",
|
|
87
|
+
"verify_compiled_config",
|
|
88
|
+
"verify_variant_execution",
|
|
89
|
+
]
|
|
90
|
+
|
|
91
|
+
#: Stable error code. Spec section 6.14.
|
|
92
|
+
VARIANT_DRY_RUN_FAILED: Final = "variant_dry_run_failed"
|
|
93
|
+
|
|
94
|
+
#: Stable error code for an execution nothing could be concluded about.
|
|
95
|
+
VARIANT_EXECUTION_UNCHECKABLE: Final = "variant_execution_uncheckable"
|
|
96
|
+
|
|
97
|
+
DEFAULT_DRY_RUN_TIMEOUT_SECONDS: Final = 300.0
|
|
98
|
+
|
|
99
|
+
#: The one key the engine fills in that Techtree deliberately left out, so a
|
|
100
|
+
#: difference there is not a disagreement. The routing header the engine may
|
|
101
|
+
#: also add needs no entry: Techtree declares no headers at all, so an added
|
|
102
|
+
#: one has nothing on the declared side to disagree with.
|
|
103
|
+
_ENGINE_RESOLVED_KEYS: Final[frozenset[str]] = frozenset({"client.base_url"})
|
|
104
|
+
|
|
105
|
+
#: Sentinel for "the key was not in the resolved document at all", which is a
|
|
106
|
+
#: different answer from "the key resolved to null" wherever null is the safe
|
|
107
|
+
#: value.
|
|
108
|
+
_MISSING: Final = object()
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
@dataclass(frozen=True)
|
|
112
|
+
class DryRunOutcome:
|
|
113
|
+
"""What one dry run established about one compiled configuration."""
|
|
114
|
+
|
|
115
|
+
variant: VariantName
|
|
116
|
+
process: EngineProcessResult
|
|
117
|
+
resolved_config_path: Path | None
|
|
118
|
+
resolved_config: dict[str, Any] | None
|
|
119
|
+
checks: tuple[ExecutionCheck, ...]
|
|
120
|
+
|
|
121
|
+
@property
|
|
122
|
+
def ok(self) -> bool:
|
|
123
|
+
"""Whether every check passed or warned."""
|
|
124
|
+
return all(check.status in ("passed", "warning") for check in self.checks)
|
|
125
|
+
|
|
126
|
+
@property
|
|
127
|
+
def failures(self) -> tuple[ExecutionCheck, ...]:
|
|
128
|
+
"""The checks that failed, in order."""
|
|
129
|
+
return tuple(check for check in self.checks if check.status == "failed")
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def dry_run_variant_config(
|
|
133
|
+
*,
|
|
134
|
+
engine_runner: EngineRunner,
|
|
135
|
+
variant: VariantName,
|
|
136
|
+
compiled: EvalToml,
|
|
137
|
+
input_config_path: Path,
|
|
138
|
+
dry_run_dir: Path,
|
|
139
|
+
model: ModelSpec | None = None,
|
|
140
|
+
timeout: float = DEFAULT_DRY_RUN_TIMEOUT_SECONDS,
|
|
141
|
+
) -> DryRunOutcome:
|
|
142
|
+
"""Resolve one compiled configuration against the installed engine.
|
|
143
|
+
|
|
144
|
+
The child is given the engine's ordinary minimal environment. A dry run
|
|
145
|
+
makes no model call, so it needs no credential, and a process that never
|
|
146
|
+
receives one cannot leak one. When ``model`` is supplied the credential is
|
|
147
|
+
*diagnosed* separately and reported as its own check, so that "the config
|
|
148
|
+
is valid" and "the endpoint can authenticate" are two answers rather than
|
|
149
|
+
one.
|
|
150
|
+
"""
|
|
151
|
+
if not input_config_path.is_file():
|
|
152
|
+
raise ValidationError(
|
|
153
|
+
"the compiled evaluation config was not written before the dry run",
|
|
154
|
+
code=VARIANT_DRY_RUN_FAILED,
|
|
155
|
+
details={"variant": variant.value, "path": str(input_config_path)},
|
|
156
|
+
)
|
|
157
|
+
ensure_private_directory(dry_run_dir)
|
|
158
|
+
|
|
159
|
+
argv = dry_run_argv(input_config_path=input_config_path, dry_run_dir=dry_run_dir)
|
|
160
|
+
process = engine_runner.run(EVAL_EXECUTABLE, argv, timeout=timeout)
|
|
161
|
+
# Spec section 6.19 keeps a record of the validation beside what it
|
|
162
|
+
# validated, so a reviewer can see exactly what was asked and answered
|
|
163
|
+
# before the run rather than inferring it from the resolved document.
|
|
164
|
+
write_command_log(
|
|
165
|
+
dry_run_dir / COMMAND_LOG_FILENAME,
|
|
166
|
+
variant=variant,
|
|
167
|
+
argv=process.argv,
|
|
168
|
+
exit_code=process.exit_code,
|
|
169
|
+
stdout=process.stdout,
|
|
170
|
+
stderr=process.stderr,
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
checks: list[ExecutionCheck] = [_invocation_check(process)]
|
|
174
|
+
resolved_path = dry_run_dir / DRY_RUN_NAME / RESOLVED_CONFIG_PATH
|
|
175
|
+
resolved: dict[str, Any] | None = None
|
|
176
|
+
|
|
177
|
+
if process.exit_code == 0 and resolved_path.is_file():
|
|
178
|
+
resolved = _read_resolved_config(resolved_path)
|
|
179
|
+
checks.append(
|
|
180
|
+
ExecutionCheck(
|
|
181
|
+
id="resolved_config_written",
|
|
182
|
+
status="passed",
|
|
183
|
+
detail=(
|
|
184
|
+
f"the engine wrote {RESOLVED_CONFIG_PATH} to the dry-run directory."
|
|
185
|
+
),
|
|
186
|
+
)
|
|
187
|
+
)
|
|
188
|
+
# ``--output-dir`` on argv overrides the file, so the resolved document
|
|
189
|
+
# names the dry-run directory rather than the real one. Comparing the
|
|
190
|
+
# compiled config as it was actually handed over keeps the check exact
|
|
191
|
+
# instead of excusing a whole key from it.
|
|
192
|
+
checks.extend(
|
|
193
|
+
verify_compiled_config(
|
|
194
|
+
compiled=compiled.model_copy(update={"output_dir": str(dry_run_dir)}),
|
|
195
|
+
resolved=resolved,
|
|
196
|
+
)
|
|
197
|
+
)
|
|
198
|
+
checks.append(_output_directory_check(compiled))
|
|
199
|
+
else:
|
|
200
|
+
checks.append(
|
|
201
|
+
ExecutionCheck(
|
|
202
|
+
id="resolved_config_written",
|
|
203
|
+
status="failed",
|
|
204
|
+
detail=(
|
|
205
|
+
"the engine wrote no resolved configuration; the dry run "
|
|
206
|
+
"did not get far enough to resolve one."
|
|
207
|
+
),
|
|
208
|
+
)
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
checks.append(_image_pinning_check(compiled))
|
|
212
|
+
if model is not None:
|
|
213
|
+
checks.append(_credential_check(model))
|
|
214
|
+
|
|
215
|
+
return DryRunOutcome(
|
|
216
|
+
variant=variant,
|
|
217
|
+
process=process,
|
|
218
|
+
resolved_config_path=resolved_path if resolved is not None else None,
|
|
219
|
+
resolved_config=resolved,
|
|
220
|
+
checks=tuple(checks),
|
|
221
|
+
)
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def verify_compiled_config(
|
|
225
|
+
*, compiled: EvalToml, resolved: Mapping[str, Any]
|
|
226
|
+
) -> list[ExecutionCheck]:
|
|
227
|
+
"""Compare what the engine resolved against what Techtree declared.
|
|
228
|
+
|
|
229
|
+
Declared means the document that was written, not the model behind it. The
|
|
230
|
+
two differ in exactly one place and it is the place that matters: ``rich``
|
|
231
|
+
is written as an explicit null, so comparing the emitted document is what
|
|
232
|
+
makes the engine confirm the dashboard is off rather than leaving that
|
|
233
|
+
unasked.
|
|
234
|
+
"""
|
|
235
|
+
declared = _flatten(emitted_document(compiled))
|
|
236
|
+
observed = _flatten(resolved)
|
|
237
|
+
|
|
238
|
+
# A declared key that is simply *missing* from the resolved document is a
|
|
239
|
+
# difference too, and it has to be spelled out rather than left to
|
|
240
|
+
# ``get``: ``rich`` is declared as a null, so an engine that resolved it
|
|
241
|
+
# into a table of dashboard settings would flatten to ``rich.show_logs``
|
|
242
|
+
# and a plain ``get`` would read the absent ``rich`` back as the null it
|
|
243
|
+
# was looking for.
|
|
244
|
+
differences = sorted(
|
|
245
|
+
key
|
|
246
|
+
for key, value in declared.items()
|
|
247
|
+
if key not in _ENGINE_RESOLVED_KEYS
|
|
248
|
+
and (key not in observed or observed[key] != value)
|
|
249
|
+
)
|
|
250
|
+
checks = [
|
|
251
|
+
ExecutionCheck(
|
|
252
|
+
id="resolved_config_matches_compiled",
|
|
253
|
+
status="passed" if not differences else "failed",
|
|
254
|
+
detail=(
|
|
255
|
+
"every value Techtree declared came back unchanged."
|
|
256
|
+
if not differences
|
|
257
|
+
else "the engine resolved a different value at "
|
|
258
|
+
f"{', '.join(differences)}."
|
|
259
|
+
),
|
|
260
|
+
),
|
|
261
|
+
_push_check(observed),
|
|
262
|
+
_serve_check(observed),
|
|
263
|
+
_subject_seat_check(observed),
|
|
264
|
+
]
|
|
265
|
+
return checks
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def _invocation_check(process: EngineProcessResult) -> ExecutionCheck:
|
|
269
|
+
"""Whether the engine's own ``eval`` accepted the configuration."""
|
|
270
|
+
if process.exit_code == 0:
|
|
271
|
+
return ExecutionCheck(
|
|
272
|
+
id="engine_eval_accepted_config",
|
|
273
|
+
status="passed",
|
|
274
|
+
detail="the engine's eval entrypoint resolved the configuration.",
|
|
275
|
+
)
|
|
276
|
+
return ExecutionCheck(
|
|
277
|
+
id="engine_eval_accepted_config",
|
|
278
|
+
status="failed",
|
|
279
|
+
detail=(
|
|
280
|
+
f"the engine's eval entrypoint exited {process.exit_code}: "
|
|
281
|
+
f"{_last_meaningful_line(process)}"
|
|
282
|
+
),
|
|
283
|
+
)
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def _serve_check(observed: Mapping[str, Any]) -> ExecutionCheck:
|
|
287
|
+
"""Whether the run the engine resolved is this process's own work.
|
|
288
|
+
|
|
289
|
+
Hosting through the elastic env-server worker pool is the default since
|
|
290
|
+
v0.3.1 (``docs/verifiers-pin-0.3.1.md``, deviation D5), and it is a default
|
|
291
|
+
Techtree cannot take: the supervision a real run depends on watches one
|
|
292
|
+
child and tears its containers down through one process group (decisions
|
|
293
|
+
document 0029). ``--no-serve`` is what turns it off, and this is the engine
|
|
294
|
+
confirming that it did rather than Techtree assuming the flag still means
|
|
295
|
+
what it meant.
|
|
296
|
+
"""
|
|
297
|
+
in_process = observed.get("serve", _MISSING) is None
|
|
298
|
+
return ExecutionCheck(
|
|
299
|
+
id="rollouts_run_in_process",
|
|
300
|
+
status="passed" if in_process else "failed",
|
|
301
|
+
detail=(
|
|
302
|
+
"the resolved configuration records no worker pool, so the "
|
|
303
|
+
"rollouts are the evaluation child's own work."
|
|
304
|
+
if in_process
|
|
305
|
+
else "the resolved configuration would host the rollouts through "
|
|
306
|
+
"an env-server worker pool rather than in the evaluation child."
|
|
307
|
+
),
|
|
308
|
+
)
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
def _push_check(observed: Mapping[str, Any]) -> ExecutionCheck:
|
|
312
|
+
"""Whether the platform upload is off in the configuration the engine read."""
|
|
313
|
+
disabled = observed.get("push") is False
|
|
314
|
+
return ExecutionCheck(
|
|
315
|
+
id="platform_push_disabled",
|
|
316
|
+
status="passed" if disabled else "failed",
|
|
317
|
+
detail=(
|
|
318
|
+
"the resolved configuration records push = false, so no episode "
|
|
319
|
+
"leaves this machine."
|
|
320
|
+
if disabled
|
|
321
|
+
else "the resolved configuration would upload this run's episodes "
|
|
322
|
+
"to the Prime platform."
|
|
323
|
+
),
|
|
324
|
+
)
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
def _subject_seat_check(observed: Mapping[str, Any]) -> ExecutionCheck:
|
|
328
|
+
"""Whether the environment the engine resolved really names the subject seat."""
|
|
329
|
+
seat_keys = [key for key in observed if key.startswith("env.subject.")]
|
|
330
|
+
agent_keys = [key for key in observed if key.startswith("env.agent.")]
|
|
331
|
+
if seat_keys and not agent_keys:
|
|
332
|
+
return ExecutionCheck(
|
|
333
|
+
id="named_subject_seat_resolved",
|
|
334
|
+
status="passed",
|
|
335
|
+
detail=(
|
|
336
|
+
"the resolved environment declares a subject seat, so every "
|
|
337
|
+
"trace will record agent.name == 'subject'."
|
|
338
|
+
),
|
|
339
|
+
)
|
|
340
|
+
return ExecutionCheck(
|
|
341
|
+
id="named_subject_seat_resolved",
|
|
342
|
+
status="failed",
|
|
343
|
+
detail=(
|
|
344
|
+
"the resolved environment does not declare a subject seat; the "
|
|
345
|
+
"reference package must export the named-subject environment."
|
|
346
|
+
),
|
|
347
|
+
)
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def _output_directory_check(compiled: EvalToml) -> ExecutionCheck:
|
|
351
|
+
"""Whether the real run would write inside the run's own directory."""
|
|
352
|
+
output_dir = Path(compiled.output_dir)
|
|
353
|
+
inside = output_dir.is_absolute() and VERIFIERS_DIRECTORY in output_dir.parts
|
|
354
|
+
return ExecutionCheck(
|
|
355
|
+
id="output_directory_is_run_owned",
|
|
356
|
+
status="passed" if inside else "failed",
|
|
357
|
+
detail=(
|
|
358
|
+
f"the real run would write to {compiled.output_dir}, inside the "
|
|
359
|
+
"run's own evaluation tree."
|
|
360
|
+
if inside
|
|
361
|
+
else f"the real run would write to {compiled.output_dir}, which is "
|
|
362
|
+
"not inside the run's own evaluation tree."
|
|
363
|
+
),
|
|
364
|
+
)
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def _image_pinning_check(compiled: EvalToml) -> ExecutionCheck:
|
|
368
|
+
"""Whether the subject runtime names content rather than a moving tag."""
|
|
369
|
+
runtime = compiled.env.subject.runtime
|
|
370
|
+
if runtime.image_is_digest_pinned:
|
|
371
|
+
return ExecutionCheck(
|
|
372
|
+
id="runtime_image_digest_pinned",
|
|
373
|
+
status="passed",
|
|
374
|
+
detail="the subject runtime image is pinned by content digest.",
|
|
375
|
+
)
|
|
376
|
+
return ExecutionCheck(
|
|
377
|
+
id="runtime_image_digest_pinned",
|
|
378
|
+
status="warning",
|
|
379
|
+
detail=(
|
|
380
|
+
f"the subject runtime image {runtime.image!r} is not pinned by "
|
|
381
|
+
"content digest, so what runs could change without the Campaign "
|
|
382
|
+
"changing."
|
|
383
|
+
),
|
|
384
|
+
)
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
def _credential_check(model: ModelSpec) -> ExecutionCheck:
|
|
388
|
+
"""Whether the evaluation endpoint can authenticate, diagnosed on its own."""
|
|
389
|
+
status = credential_status(model)
|
|
390
|
+
return ExecutionCheck(
|
|
391
|
+
id="evaluation_credential_available",
|
|
392
|
+
status="passed" if status.available else "failed",
|
|
393
|
+
detail=status.detail,
|
|
394
|
+
)
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
def _read_resolved_config(path: Path) -> dict[str, Any]:
|
|
398
|
+
"""Read the configuration the engine wrote back."""
|
|
399
|
+
try:
|
|
400
|
+
document: dict[str, Any] = json.loads(path.read_text(encoding="utf-8"))
|
|
401
|
+
return document
|
|
402
|
+
except (OSError, UnicodeDecodeError, json.JSONDecodeError) as error:
|
|
403
|
+
raise ValidationError(
|
|
404
|
+
"the engine's resolved configuration could not be read",
|
|
405
|
+
code=VARIANT_DRY_RUN_FAILED,
|
|
406
|
+
details={"path": str(path)},
|
|
407
|
+
) from error
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
def _flatten(document: Mapping[str, Any], prefix: str = "") -> dict[str, Any]:
|
|
411
|
+
"""Flatten nested tables into dotted keys so two documents can be compared."""
|
|
412
|
+
flat: dict[str, Any] = {}
|
|
413
|
+
for key, value in document.items():
|
|
414
|
+
dotted = f"{prefix}{key}"
|
|
415
|
+
if isinstance(value, dict):
|
|
416
|
+
flat.update(_flatten(value, prefix=f"{dotted}."))
|
|
417
|
+
else:
|
|
418
|
+
flat[dotted] = value
|
|
419
|
+
return flat
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
def _last_meaningful_line(process: EngineProcessResult) -> str:
|
|
423
|
+
"""Return the most useful line of a failed invocation's output.
|
|
424
|
+
|
|
425
|
+
The engine renders configuration errors as a box-drawn panel, so the raw
|
|
426
|
+
tail is mostly border. This keeps the last line that carries characters
|
|
427
|
+
other than the frame.
|
|
428
|
+
"""
|
|
429
|
+
frame = set("│╭╮╰╯─ ")
|
|
430
|
+
for stream in (process.stderr, process.stdout):
|
|
431
|
+
lines = [
|
|
432
|
+
line.strip("│ ").strip()
|
|
433
|
+
for line in reversed(stream.splitlines())
|
|
434
|
+
if line.strip() and set(line) - frame
|
|
435
|
+
]
|
|
436
|
+
if lines:
|
|
437
|
+
return lines[0]
|
|
438
|
+
return "<no output>"
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
# ---------------------------------------------------------------------------
|
|
442
|
+
# After the run: is this execution scientifically usable
|
|
443
|
+
# ---------------------------------------------------------------------------
|
|
444
|
+
|
|
445
|
+
|
|
446
|
+
def verify_variant_execution(
|
|
447
|
+
*,
|
|
448
|
+
result: VariantExecutionResult,
|
|
449
|
+
experiment: ExperimentManifest,
|
|
450
|
+
taskset_lock: TasksetLock,
|
|
451
|
+
primary_reward: str,
|
|
452
|
+
engine: EngineDescriptor | None = None,
|
|
453
|
+
) -> list[ExecutionCheck]:
|
|
454
|
+
"""Return ordered checks over one completed variant.
|
|
455
|
+
|
|
456
|
+
Raises only when the inputs cannot be checked at all — a manifest with no
|
|
457
|
+
subject agent leaves nothing to compare an execution against. Everything
|
|
458
|
+
else is reported as a verdict, because "which rule did this break" is the
|
|
459
|
+
question a caller actually has.
|
|
460
|
+
"""
|
|
461
|
+
subject = experiment.configuration.agents.get(SUBJECT_AGENT)
|
|
462
|
+
if subject is None:
|
|
463
|
+
raise ValidationError(
|
|
464
|
+
f"the experiment manifest defines no {SUBJECT_AGENT!r} agent, so "
|
|
465
|
+
"there is nothing to verify this execution against",
|
|
466
|
+
code=VARIANT_EXECUTION_UNCHECKABLE,
|
|
467
|
+
details={"manifest_id": experiment.id, "variant": result.variant.value},
|
|
468
|
+
)
|
|
469
|
+
|
|
470
|
+
checks = [_completion_check(result)]
|
|
471
|
+
checks.extend(_membership_checks(result, taskset_lock))
|
|
472
|
+
checks.extend(_trace_checks(result, subject, primary_reward))
|
|
473
|
+
checks.append(_manifest_check(result, experiment))
|
|
474
|
+
if engine is not None:
|
|
475
|
+
checks.append(_pin_check(result, engine))
|
|
476
|
+
return checks
|
|
477
|
+
|
|
478
|
+
|
|
479
|
+
def _completion_check(result: VariantExecutionResult) -> ExecutionCheck:
|
|
480
|
+
"""Whether the child finished rather than being stopped."""
|
|
481
|
+
outcome = result.child_outcome
|
|
482
|
+
if outcome.cancelled or outcome.exit_code == CANCELLATION_EXIT_CODE:
|
|
483
|
+
return ExecutionCheck(
|
|
484
|
+
id="child_completed",
|
|
485
|
+
status="failed",
|
|
486
|
+
detail=(
|
|
487
|
+
"the evaluation was cancelled, so it produced no answer rather "
|
|
488
|
+
"than a wrong one."
|
|
489
|
+
),
|
|
490
|
+
)
|
|
491
|
+
if outcome.exit_code != 0:
|
|
492
|
+
return ExecutionCheck(
|
|
493
|
+
id="child_completed",
|
|
494
|
+
status="failed",
|
|
495
|
+
detail=f"the evaluation child exited {outcome.exit_code}.",
|
|
496
|
+
)
|
|
497
|
+
return ExecutionCheck(
|
|
498
|
+
id="child_completed",
|
|
499
|
+
status="passed",
|
|
500
|
+
detail="the evaluation child ran to completion and was not cancelled.",
|
|
501
|
+
)
|
|
502
|
+
|
|
503
|
+
|
|
504
|
+
def _membership_checks(
|
|
505
|
+
result: VariantExecutionResult, taskset_lock: TasksetLock
|
|
506
|
+
) -> list[ExecutionCheck]:
|
|
507
|
+
"""Whether exactly the committed tasks were scored, in the committed order."""
|
|
508
|
+
observed = [episode.task_hash for episode in result.episodes]
|
|
509
|
+
committed = list(taskset_lock.ordered_task_hashes)
|
|
510
|
+
|
|
511
|
+
count = ExecutionCheck(
|
|
512
|
+
id="episode_count",
|
|
513
|
+
status="passed" if len(observed) == len(committed) else "failed",
|
|
514
|
+
detail=(
|
|
515
|
+
f"{len(observed)} episodes for {len(committed)} committed tasks."
|
|
516
|
+
if len(observed) == len(committed)
|
|
517
|
+
else f"{len(observed)} episodes were recorded for {len(committed)} "
|
|
518
|
+
"committed tasks."
|
|
519
|
+
),
|
|
520
|
+
)
|
|
521
|
+
ordered = ExecutionCheck(
|
|
522
|
+
id="ordered_task_membership",
|
|
523
|
+
status="passed" if observed == committed else "failed",
|
|
524
|
+
detail=(
|
|
525
|
+
"the normalized episodes cover exactly the committed tasks, in the "
|
|
526
|
+
"committed order."
|
|
527
|
+
if observed == committed
|
|
528
|
+
else "the normalized episodes do not match the committed membership; "
|
|
529
|
+
"pairing joins on task hash, never on position."
|
|
530
|
+
),
|
|
531
|
+
)
|
|
532
|
+
positions = [episode.task_position for episode in result.episodes]
|
|
533
|
+
numbered = ExecutionCheck(
|
|
534
|
+
id="task_positions_are_membership_positions",
|
|
535
|
+
status="passed" if positions == list(range(len(observed))) else "failed",
|
|
536
|
+
detail=(
|
|
537
|
+
"every episode carries its membership position."
|
|
538
|
+
if positions == list(range(len(observed)))
|
|
539
|
+
else f"episode positions are not 0..{len(observed) - 1} exactly once."
|
|
540
|
+
),
|
|
541
|
+
)
|
|
542
|
+
|
|
543
|
+
episode_ids = [episode.episode_id for episode in result.episodes]
|
|
544
|
+
unique = ExecutionCheck(
|
|
545
|
+
id="episode_ids_unique",
|
|
546
|
+
status="passed" if len(set(episode_ids)) == len(episode_ids) else "failed",
|
|
547
|
+
detail=(
|
|
548
|
+
"every episode has its own identifier."
|
|
549
|
+
if len(set(episode_ids)) == len(episode_ids)
|
|
550
|
+
else "two episodes share an identifier."
|
|
551
|
+
),
|
|
552
|
+
)
|
|
553
|
+
completed = [episode for episode in result.episodes if not episode.ok]
|
|
554
|
+
finished = ExecutionCheck(
|
|
555
|
+
id="all_episodes_completed",
|
|
556
|
+
status="passed" if not completed else "failed",
|
|
557
|
+
detail=(
|
|
558
|
+
"every episode completed."
|
|
559
|
+
if not completed
|
|
560
|
+
else f"{len(completed)} episode(s) did not complete."
|
|
561
|
+
),
|
|
562
|
+
)
|
|
563
|
+
return [count, ordered, numbered, unique, finished]
|
|
564
|
+
|
|
565
|
+
|
|
566
|
+
def _trace_checks(
|
|
567
|
+
result: VariantExecutionResult, subject: AgentSpec, primary_reward: str
|
|
568
|
+
) -> list[ExecutionCheck]:
|
|
569
|
+
"""Whether every trace is the subject the manifest declared, scored."""
|
|
570
|
+
traces = [trace for episode in result.episodes for trace in episode.traces]
|
|
571
|
+
per_episode = {len(episode.traces) for episode in result.episodes}
|
|
572
|
+
|
|
573
|
+
checks = [
|
|
574
|
+
ExecutionCheck(
|
|
575
|
+
id="one_subject_trace_per_episode",
|
|
576
|
+
status="passed" if per_episode <= {1} else "failed",
|
|
577
|
+
detail=(
|
|
578
|
+
"each episode carries exactly one subject trace."
|
|
579
|
+
if per_episode <= {1}
|
|
580
|
+
else f"episodes carry {sorted(per_episode)} traces."
|
|
581
|
+
),
|
|
582
|
+
)
|
|
583
|
+
]
|
|
584
|
+
trace_ids = [trace.trace_id for trace in traces]
|
|
585
|
+
checks.append(
|
|
586
|
+
ExecutionCheck(
|
|
587
|
+
id="trace_ids_unique",
|
|
588
|
+
status="passed" if len(set(trace_ids)) == len(trace_ids) else "failed",
|
|
589
|
+
detail=(
|
|
590
|
+
"every trace has its own identifier."
|
|
591
|
+
if len(set(trace_ids)) == len(trace_ids)
|
|
592
|
+
else "two traces share an identifier."
|
|
593
|
+
),
|
|
594
|
+
)
|
|
595
|
+
)
|
|
596
|
+
|
|
597
|
+
declared_skills = sorted(artifact.digest for artifact in subject.harness.skills)
|
|
598
|
+
expectations: list[tuple[str, str, Any, Any]] = [
|
|
599
|
+
("model_id_matches", "model", subject.model.model_id, None),
|
|
600
|
+
("harness_id_matches", "harness", subject.harness.id, None),
|
|
601
|
+
("harness_version_matches", "harness version", subject.harness.version, None),
|
|
602
|
+
("runtime_image_matches", "runtime image", subject.runtime.image, None),
|
|
603
|
+
]
|
|
604
|
+
observed: dict[str, set[Any]] = {
|
|
605
|
+
"model_id_matches": {trace.model_id for trace in traces},
|
|
606
|
+
"harness_id_matches": {trace.harness_id for trace in traces},
|
|
607
|
+
"harness_version_matches": {trace.harness_version for trace in traces},
|
|
608
|
+
"runtime_image_matches": {trace.runtime.image for trace in traces},
|
|
609
|
+
}
|
|
610
|
+
for identifier, label, expected, _ in expectations:
|
|
611
|
+
seen = observed[identifier]
|
|
612
|
+
checks.append(
|
|
613
|
+
ExecutionCheck(
|
|
614
|
+
id=identifier,
|
|
615
|
+
status="passed" if seen == {expected} else "failed",
|
|
616
|
+
detail=(
|
|
617
|
+
f"every trace ran the declared {label}."
|
|
618
|
+
if seen == {expected}
|
|
619
|
+
else f"the declared {label} is {expected!r}; traces recorded "
|
|
620
|
+
f"{sorted(str(value) for value in seen)}."
|
|
621
|
+
),
|
|
622
|
+
)
|
|
623
|
+
)
|
|
624
|
+
|
|
625
|
+
bundled = {trace.use_bundled_skill for trace in traces}
|
|
626
|
+
checks.append(
|
|
627
|
+
ExecutionCheck(
|
|
628
|
+
id="bundled_skills_disabled",
|
|
629
|
+
status="passed" if bundled <= {False} else "failed",
|
|
630
|
+
detail=(
|
|
631
|
+
"no trace enabled the harness's bundled skill catalogue."
|
|
632
|
+
if bundled <= {False}
|
|
633
|
+
else "a trace ran with the bundled skill catalogue enabled."
|
|
634
|
+
),
|
|
635
|
+
)
|
|
636
|
+
)
|
|
637
|
+
skills = {tuple(sorted(trace.skill_root_digests)) for trace in traces}
|
|
638
|
+
checks.append(
|
|
639
|
+
ExecutionCheck(
|
|
640
|
+
id="skill_digests_match_variant",
|
|
641
|
+
status="passed" if skills <= {tuple(declared_skills)} else "failed",
|
|
642
|
+
detail=(
|
|
643
|
+
f"every trace carried the {len(declared_skills)} skill(s) this "
|
|
644
|
+
"variant declares."
|
|
645
|
+
if skills <= {tuple(declared_skills)}
|
|
646
|
+
else "a trace carried skills the variant does not declare."
|
|
647
|
+
),
|
|
648
|
+
)
|
|
649
|
+
)
|
|
650
|
+
runtimes = {trace.runtime.kind for trace in traces}
|
|
651
|
+
checks.append(
|
|
652
|
+
ExecutionCheck(
|
|
653
|
+
id="runtime_is_docker",
|
|
654
|
+
status="passed" if runtimes <= {"docker"} else "failed",
|
|
655
|
+
detail=(
|
|
656
|
+
"every trace ran in a Docker runtime."
|
|
657
|
+
if runtimes <= {"docker"}
|
|
658
|
+
else f"traces ran on {sorted(runtimes)}."
|
|
659
|
+
),
|
|
660
|
+
)
|
|
661
|
+
)
|
|
662
|
+
roles = {trace.agent_role for trace in traces}
|
|
663
|
+
checks.append(
|
|
664
|
+
ExecutionCheck(
|
|
665
|
+
id="every_trace_is_the_subject",
|
|
666
|
+
status="passed" if roles <= {SUBJECT_AGENT} else "failed",
|
|
667
|
+
detail=(
|
|
668
|
+
f"every trace records the {SUBJECT_AGENT!r} role."
|
|
669
|
+
if roles <= {SUBJECT_AGENT}
|
|
670
|
+
else f"traces recorded the roles {sorted(roles)}."
|
|
671
|
+
),
|
|
672
|
+
)
|
|
673
|
+
)
|
|
674
|
+
incomplete = [trace for trace in traces if not trace.ok]
|
|
675
|
+
checks.append(
|
|
676
|
+
ExecutionCheck(
|
|
677
|
+
id="all_traces_completed",
|
|
678
|
+
status="passed" if not incomplete else "failed",
|
|
679
|
+
detail=(
|
|
680
|
+
"every trace completed."
|
|
681
|
+
if not incomplete
|
|
682
|
+
else f"{len(incomplete)} trace(s) did not complete."
|
|
683
|
+
),
|
|
684
|
+
)
|
|
685
|
+
)
|
|
686
|
+
checks.append(_primary_reward_check(traces, primary_reward))
|
|
687
|
+
checks.append(_tool_inventory_check(traces))
|
|
688
|
+
return checks
|
|
689
|
+
|
|
690
|
+
|
|
691
|
+
def _primary_reward_check(
|
|
692
|
+
traces: list[NormalizedTrace], primary_reward: str
|
|
693
|
+
) -> ExecutionCheck:
|
|
694
|
+
"""Whether the reward the comparison turns on was scored everywhere."""
|
|
695
|
+
unscored = [trace for trace in traces if trace.reward(primary_reward) is None]
|
|
696
|
+
if unscored:
|
|
697
|
+
return ExecutionCheck(
|
|
698
|
+
id="primary_reward_present",
|
|
699
|
+
status="failed",
|
|
700
|
+
detail=(
|
|
701
|
+
f"{len(unscored)} trace(s) carry no {primary_reward!r} reward, "
|
|
702
|
+
"which is the reward the comparison is decided on."
|
|
703
|
+
),
|
|
704
|
+
)
|
|
705
|
+
return ExecutionCheck(
|
|
706
|
+
id="primary_reward_present",
|
|
707
|
+
status="passed",
|
|
708
|
+
detail=f"every trace scored {primary_reward!r}.",
|
|
709
|
+
)
|
|
710
|
+
|
|
711
|
+
|
|
712
|
+
def _tool_inventory_check(traces: list[NormalizedTrace]) -> ExecutionCheck:
|
|
713
|
+
"""Whether each trace's recorded tool inventory is internally coherent."""
|
|
714
|
+
for trace in traces:
|
|
715
|
+
names = [tool.name for tool in trace.tools]
|
|
716
|
+
if len(set(names)) != len(names):
|
|
717
|
+
return ExecutionCheck(
|
|
718
|
+
id="tool_inventory_valid",
|
|
719
|
+
status="failed",
|
|
720
|
+
detail=f"trace {trace.trace_id} advertises a tool name twice.",
|
|
721
|
+
)
|
|
722
|
+
return ExecutionCheck(
|
|
723
|
+
id="tool_inventory_valid",
|
|
724
|
+
status="passed",
|
|
725
|
+
detail="every trace's tool inventory names each tool once.",
|
|
726
|
+
)
|
|
727
|
+
|
|
728
|
+
|
|
729
|
+
def _manifest_check(
|
|
730
|
+
result: VariantExecutionResult, experiment: ExperimentManifest
|
|
731
|
+
) -> ExecutionCheck:
|
|
732
|
+
"""Whether this result belongs to the manifest it is being checked against."""
|
|
733
|
+
from techtree.canonical import digest_object
|
|
734
|
+
|
|
735
|
+
expected = digest_object(experiment)
|
|
736
|
+
matches = result.experiment_manifest_digest == expected
|
|
737
|
+
return ExecutionCheck(
|
|
738
|
+
id="result_matches_experiment",
|
|
739
|
+
status="passed" if matches else "failed",
|
|
740
|
+
detail=(
|
|
741
|
+
"the result was produced from this experiment manifest."
|
|
742
|
+
if matches
|
|
743
|
+
else "the result names a different experiment manifest."
|
|
744
|
+
),
|
|
745
|
+
)
|
|
746
|
+
|
|
747
|
+
|
|
748
|
+
def _pin_check(
|
|
749
|
+
result: VariantExecutionResult, engine: EngineDescriptor
|
|
750
|
+
) -> ExecutionCheck:
|
|
751
|
+
"""Whether the build that produced this result is the build we pinned.
|
|
752
|
+
|
|
753
|
+
Nothing is inferred here. Every upstream trace records the Verifiers
|
|
754
|
+
version and commit that wrote it (``docs/verifiers-eval.md``) and the
|
|
755
|
+
normalizer carries both across, so this compares evidence against the
|
|
756
|
+
engine descriptor rather than trusting a caller's claim about which engine
|
|
757
|
+
ran.
|
|
758
|
+
"""
|
|
759
|
+
observed = {
|
|
760
|
+
(trace.verifiers_version, trace.verifiers_revision)
|
|
761
|
+
for episode in result.episodes
|
|
762
|
+
for trace in episode.traces
|
|
763
|
+
}
|
|
764
|
+
expected = (engine.verifiers_version, engine.verifiers_revision)
|
|
765
|
+
if observed == {expected}:
|
|
766
|
+
return ExecutionCheck(
|
|
767
|
+
id="verifiers_pin_matches_engine",
|
|
768
|
+
status="passed",
|
|
769
|
+
detail=(
|
|
770
|
+
f"every trace was produced by Verifiers {expected[0]} at "
|
|
771
|
+
f"{expected[1]}, which is what the engine descriptor pins."
|
|
772
|
+
),
|
|
773
|
+
)
|
|
774
|
+
return ExecutionCheck(
|
|
775
|
+
id="verifiers_pin_matches_engine",
|
|
776
|
+
status="failed",
|
|
777
|
+
detail=(
|
|
778
|
+
f"the engine descriptor pins Verifiers {expected[0]} at "
|
|
779
|
+
f"{expected[1]}, but the traces record "
|
|
780
|
+
f"{sorted(f'{version} at {revision}' for version, revision in observed)}."
|
|
781
|
+
),
|
|
782
|
+
)
|