techtree 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- techtree/__init__.py +35 -0
- techtree/__main__.py +14 -0
- techtree/canonical.py +239 -0
- techtree/catalog/__init__.py +25 -0
- techtree/catalog/repository.py +400 -0
- techtree/catalog/service.py +419 -0
- techtree/cli/__init__.py +1 -0
- techtree/cli/app.py +416 -0
- techtree/cli/commands/__init__.py +1 -0
- techtree/cli/commands/climb.py +1223 -0
- techtree/cli/commands/doctor.py +147 -0
- techtree/cli/commands/engine.py +207 -0
- techtree/cli/commands/proof.py +556 -0
- techtree/cli/commands/publish.py +447 -0
- techtree/cli/commands/release.py +303 -0
- techtree/cli/commands/run.py +1067 -0
- techtree/cli/commands/setup.py +181 -0
- techtree/cli/commands/skill.py +221 -0
- techtree/cli/commands/uplift.py +698 -0
- techtree/cli/commands/withdraw.py +212 -0
- techtree/cli/confirm.py +47 -0
- techtree/cli/context.py +96 -0
- techtree/cli/invoke.py +220 -0
- techtree/cli/output.py +280 -0
- techtree/constants.py +138 -0
- techtree/crypto.py +128 -0
- techtree/doctor/__init__.py +1 -0
- techtree/doctor/checks.py +675 -0
- techtree/doctor/execution_checks.py +435 -0
- techtree/doctor/service.py +326 -0
- techtree/drafts/__init__.py +32 -0
- techtree/drafts/source.py +146 -0
- techtree/drafts/store.py +992 -0
- techtree/engines/__init__.py +1 -0
- techtree/engines/bundle.py +251 -0
- techtree/engines/installer.py +679 -0
- techtree/engines/registry.py +235 -0
- techtree/engines/runner.py +170 -0
- techtree/errors.py +262 -0
- techtree/fs.py +234 -0
- techtree/harness.py +108 -0
- techtree/identity/__init__.py +41 -0
- techtree/identity/models.py +113 -0
- techtree/identity/service.py +199 -0
- techtree/identity/store.py +263 -0
- techtree/ids.py +85 -0
- techtree/manifests/__init__.py +39 -0
- techtree/manifests/builder.py +433 -0
- techtree/manifests/compare.py +376 -0
- techtree/models/__init__.py +282 -0
- techtree/models/base.py +201 -0
- techtree/models/campaign.py +484 -0
- techtree/models/catalog.py +227 -0
- techtree/models/cli.py +151 -0
- techtree/models/climb.py +254 -0
- techtree/models/data_policy.py +130 -0
- techtree/models/engine.py +156 -0
- techtree/models/episode_receipt.py +130 -0
- techtree/models/evaluation_backend.py +113 -0
- techtree/models/experiment.py +154 -0
- techtree/models/run.py +214 -0
- techtree/models/skill.py +156 -0
- techtree/models/uplift_report.py +158 -0
- techtree/models/validation.py +299 -0
- techtree/paths.py +116 -0
- techtree/presentation/__init__.py +31 -0
- techtree/presentation/build.py +1242 -0
- techtree/presentation/compact.py +246 -0
- techtree/presentation/evidence.py +169 -0
- techtree/presentation/models.py +358 -0
- techtree/presentation/rich.py +312 -0
- techtree/presentation/sanitize.py +156 -0
- techtree/publication/__init__.py +44 -0
- techtree/publication/address.py +180 -0
- techtree/publication/coordinates.py +26 -0
- techtree/publication/journal.py +212 -0
- techtree/publication/keccak.py +183 -0
- techtree/publication/models.py +209 -0
- techtree/publication/offer.py +35 -0
- techtree/publication/service.py +618 -0
- techtree/publication/transport.py +296 -0
- techtree/publication/verify.py +242 -0
- techtree/publication/withdraw.py +156 -0
- techtree/py.typed +0 -0
- techtree/receipts/__init__.py +52 -0
- techtree/receipts/bundle.py +578 -0
- techtree/receipts/compare.py +1065 -0
- techtree/receipts/episode.py +672 -0
- techtree/receipts/execution.py +630 -0
- techtree/receipts/observed.py +474 -0
- techtree/receipts/set.py +336 -0
- techtree/receipts/uplift.py +655 -0
- techtree/receipts/verify.py +1055 -0
- techtree/release/__init__.py +9 -0
- techtree/release/bootstrap.py +509 -0
- techtree/release/checks.py +376 -0
- techtree/release/document.py +125 -0
- techtree/release/generate.py +221 -0
- techtree/release/models.py +293 -0
- techtree/release/provenance.py +109 -0
- techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
- techtree/resources/catalog/catalog.json +32 -0
- techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
- techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
- techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
- techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
- techtree/resources/engines/default/engine.json +20 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
- techtree/resources/engines/default/pyproject.toml +23 -0
- techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
- techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
- techtree/resources/engines/default/tools/normalize_validation.py +222 -0
- techtree/resources/engines/default/uv.lock +1758 -0
- techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
- techtree/resources/release/build-provenance.json +4 -0
- techtree/resources/release/release-core.json +24 -0
- techtree/runs/__init__.py +31 -0
- techtree/runs/artifacts.py +750 -0
- techtree/runs/child_registry.py +228 -0
- techtree/runs/events.py +478 -0
- techtree/runs/executor.py +140 -0
- techtree/runs/fake.py +741 -0
- techtree/runs/launcher.py +253 -0
- techtree/runs/machine.py +489 -0
- techtree/runs/real.py +789 -0
- techtree/runs/service.py +616 -0
- techtree/runs/store.py +555 -0
- techtree/runs/validation.py +259 -0
- techtree/runs/variants.py +684 -0
- techtree/settings.py +143 -0
- techtree/skills/__init__.py +14 -0
- techtree/skills/archive.py +282 -0
- techtree/skills/policy.py +62 -0
- techtree/skills/scanner.py +394 -0
- techtree/skills/service.py +752 -0
- techtree/skills/starter.py +434 -0
- techtree/tasksets/__init__.py +1 -0
- techtree/tasksets/membership.py +269 -0
- techtree/tasksets/provider.py +207 -0
- techtree/tasksets/resolver.py +311 -0
- techtree/tasksets/service.py +484 -0
- techtree/tasksets/verifiers_cli.py +538 -0
- techtree/uplift/__init__.py +20 -0
- techtree/uplift/context.py +544 -0
- techtree/uplift/derive.py +203 -0
- techtree/uplift/public_tasks.py +151 -0
- techtree/uplift/service.py +719 -0
- techtree/uplift/source.py +160 -0
- techtree/verifiers/__init__.py +31 -0
- techtree/verifiers/budget.py +219 -0
- techtree/verifiers/child.py +633 -0
- techtree/verifiers/compiler.py +432 -0
- techtree/verifiers/config.py +365 -0
- techtree/verifiers/credentials.py +321 -0
- techtree/verifiers/image.py +126 -0
- techtree/verifiers/models.py +527 -0
- techtree/verifiers/outputs.py +368 -0
- techtree/verifiers/progress.py +192 -0
- techtree/verifiers/supervisor.py +341 -0
- techtree/verifiers/verify.py +782 -0
- techtree/version.py +39 -0
- techtree/worker/__init__.py +18 -0
- techtree/worker/execute.py +487 -0
- techtree/worker/main.py +57 -0
- techtree-0.1.0.dist-info/METADATA +344 -0
- techtree-0.1.0.dist-info/RECORD +174 -0
- techtree-0.1.0.dist-info/WHEEL +4 -0
- techtree-0.1.0.dist-info/entry_points.txt +3 -0
- techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,1065 @@
|
|
|
1
|
+
"""Proving the two variants were one experiment. Spec section 7.9.
|
|
2
|
+
|
|
3
|
+
:mod:`techtree.manifests.compare` proves that the two documents a run was
|
|
4
|
+
prepared from differ only where the Campaign permits. That is a claim about
|
|
5
|
+
*intentions*. This module makes the harder claim: that the two executions the
|
|
6
|
+
run actually performed differ only there too.
|
|
7
|
+
|
|
8
|
+
The difference matters because everything between a manifest and a container
|
|
9
|
+
can drift. A provider can route a model identifier somewhere else, a harness can
|
|
10
|
+
resolve a different version, a daemon can hand back a different image, a
|
|
11
|
+
configuration can pick up a sampling parameter nobody declared. None of that is
|
|
12
|
+
visible in a manifest, and all of it would be reported as uplift.
|
|
13
|
+
|
|
14
|
+
So the comparison runs over three sources at once:
|
|
15
|
+
|
|
16
|
+
*What was declared* — the two ``ExperimentManifest`` documents and the Campaign
|
|
17
|
+
they were derived from, checked field by field and then diffed whole.
|
|
18
|
+
|
|
19
|
+
*What was observed* — one fingerprint per variant, computed by
|
|
20
|
+
:mod:`techtree.receipts.observed` from the traces the runtime wrote and the
|
|
21
|
+
configuration the engine resolved, checked against the manifest that variant was
|
|
22
|
+
supposed to be and against the other variant's fingerprint.
|
|
23
|
+
|
|
24
|
+
*What was joined* — one receipt per committed task on each side, paired by task
|
|
25
|
+
hash in the order the TasksetLock fixes, so the aggregation downstream cannot
|
|
26
|
+
pair by arrival order or quietly drop a task.
|
|
27
|
+
|
|
28
|
+
Four decisions are worth stating.
|
|
29
|
+
|
|
30
|
+
*The whole tool surface cannot be required to match, and here is exactly what
|
|
31
|
+
may not.* Mounting a Skill changes what the subject is offered, because Hermes
|
|
32
|
+
0.19.0 renders the index of visible Skills into the description of its own
|
|
33
|
+
Skill-management tool. Across two recorded probes of the same Campaign the
|
|
34
|
+
fifteen tool names, fifteen parameter schemas and fourteen of the fifteen
|
|
35
|
+
descriptions are byte-identical, and ``skill_manage``'s description differs. A
|
|
36
|
+
check that required one tool-inventory digest would therefore reject every
|
|
37
|
+
clean run. What is permitted is the exact derived difference decisions document
|
|
38
|
+
0007 ratified: the tool names and parameter schemas of the pinned harness
|
|
39
|
+
conformance fixture (:mod:`techtree.harness`), unchanged on both sides, with at
|
|
40
|
+
most one differing description and only on :data:`SKILL_INDEX_TOOL`. A second
|
|
41
|
+
differing description, a differing schema, a differing description anywhere
|
|
42
|
+
else, or any departure from the fixture's own surface is a violation — the
|
|
43
|
+
fixture is what stops a moved harness pin from being read as the Skill index
|
|
44
|
+
doing what the Skill index does.
|
|
45
|
+
|
|
46
|
+
*A weaker claim is a warning, never silence and never a failure.* One thing
|
|
47
|
+
about a real run is honestly unverifiable here: a provider that publishes no
|
|
48
|
+
model revision cannot be pinned to one. It is recorded as a warning, which is
|
|
49
|
+
what
|
|
50
|
+
:class:`~techtree.models.uplift_report.ComparisonStatus.CONTROLLED_WITH_WARNINGS`
|
|
51
|
+
exists for, and it is never allowed to absorb an actual mismatch: a model that
|
|
52
|
+
differs is a failure whether or not its revision was discoverable. What ran in
|
|
53
|
+
the container used to be a warning of the same kind and is a check now — see
|
|
54
|
+
:func:`_runtime_pin_checks`.
|
|
55
|
+
|
|
56
|
+
*This function reports; it does not refuse.* Like
|
|
57
|
+
:func:`~techtree.manifests.compare.compare_manifests`, it returns what it found
|
|
58
|
+
so that an operator can read every check. Turning an invalid comparison into a
|
|
59
|
+
refusal is :mod:`techtree.receipts.uplift`'s job, at the point where a report
|
|
60
|
+
would otherwise be written.
|
|
61
|
+
|
|
62
|
+
*The observed inputs are richer than section 7.9's signature.* The specification
|
|
63
|
+
sketches ``compare_real_variants`` as taking receipts alone. A frozen
|
|
64
|
+
``EpisodeReceipt`` carries rewards and lineage and no configuration at all, so
|
|
65
|
+
the observed side is passed explicitly as :class:`ObservedVariant`, built by
|
|
66
|
+
:func:`observe_variant` from the same evidence the receipts were built from.
|
|
67
|
+
The schedule is passed for the same reason: which schedule ran, and how far
|
|
68
|
+
apart the two children were, is a fact about the execution rather than about
|
|
69
|
+
either variant.
|
|
70
|
+
"""
|
|
71
|
+
|
|
72
|
+
from __future__ import annotations
|
|
73
|
+
|
|
74
|
+
from collections.abc import Mapping, Sequence
|
|
75
|
+
from dataclasses import dataclass
|
|
76
|
+
from typing import Any, Final, Literal, Self
|
|
77
|
+
|
|
78
|
+
from pydantic import Field, model_validator
|
|
79
|
+
|
|
80
|
+
from techtree.canonical import digest_object, validate_digest
|
|
81
|
+
from techtree.harness import harness_conformance
|
|
82
|
+
from techtree.manifests.compare import compare_manifests
|
|
83
|
+
from techtree.models.base import Digest, JsonValue, NonEmptyString, ProtocolModel
|
|
84
|
+
from techtree.models.campaign import (
|
|
85
|
+
SUBJECT_AGENT,
|
|
86
|
+
AgentSpec,
|
|
87
|
+
CampaignSpec,
|
|
88
|
+
MutationKind,
|
|
89
|
+
RuntimeSpec,
|
|
90
|
+
VariantSchedule,
|
|
91
|
+
)
|
|
92
|
+
from techtree.models.episode_receipt import EpisodeReceipt
|
|
93
|
+
from techtree.models.experiment import (
|
|
94
|
+
ExperimentManifest,
|
|
95
|
+
ExperimentVariant,
|
|
96
|
+
ManifestComparison,
|
|
97
|
+
)
|
|
98
|
+
from techtree.models.uplift_report import ComparisonStatus
|
|
99
|
+
from techtree.models.validation import TasksetLock
|
|
100
|
+
from techtree.receipts.observed import (
|
|
101
|
+
ObservedSubjectConfiguration,
|
|
102
|
+
observed_from_episodes,
|
|
103
|
+
)
|
|
104
|
+
from techtree.tasksets.membership import membership_digest
|
|
105
|
+
from techtree.verifiers.models import (
|
|
106
|
+
ChildProcessOutcome,
|
|
107
|
+
NormalizedTool,
|
|
108
|
+
VariantExecutionResult,
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
__all__ = [
|
|
112
|
+
"COMPARISON_INVALID",
|
|
113
|
+
"MODEL_REVISION_UNDISCOVERABLE",
|
|
114
|
+
"SKILL_INDEX_TOOL",
|
|
115
|
+
"ComparisonCheck",
|
|
116
|
+
"ObservedVariant",
|
|
117
|
+
"PairedReceiptRow",
|
|
118
|
+
"RealComparisonResult",
|
|
119
|
+
"ScheduleObservation",
|
|
120
|
+
"compare_real_variants",
|
|
121
|
+
"observe_variant",
|
|
122
|
+
"weaker_claim_warnings",
|
|
123
|
+
]
|
|
124
|
+
|
|
125
|
+
#: Stable error code. Spec section 15. Raised by whoever turns an invalid
|
|
126
|
+
#: comparison into a refusal, never by this module.
|
|
127
|
+
COMPARISON_INVALID: Final = "comparison_invalid"
|
|
128
|
+
|
|
129
|
+
#: The one tool whose *description* two controlled variants may disagree about.
|
|
130
|
+
#:
|
|
131
|
+
#: Hermes 0.19.0 lists the Skills currently visible to the subject inside this
|
|
132
|
+
#: tool's description, so inserting or replacing a Skill necessarily changes it.
|
|
133
|
+
#: It is a consequence of the mutation rather than a second difference, and it
|
|
134
|
+
#: was measured on the recorded probes rather than assumed: see
|
|
135
|
+
#: ``tests/fixtures/receipts/support.py`` and
|
|
136
|
+
#: ``test_inserting_a_skill_changes_exactly_one_tool_description``.
|
|
137
|
+
SKILL_INDEX_TOOL: Final = "skill_manage"
|
|
138
|
+
|
|
139
|
+
#: The one weaker-claim warning this build can record, named so that a reader
|
|
140
|
+
#: of a result can be told which coordinate it is about rather than only that
|
|
141
|
+
#: there was one. :func:`weaker_claim_warnings` is what decides whether a run
|
|
142
|
+
#: has it.
|
|
143
|
+
MODEL_REVISION_UNDISCOVERABLE: Final = "model_revision_discoverable"
|
|
144
|
+
|
|
145
|
+
_PASSED: Final = "passed"
|
|
146
|
+
_FAILED: Final = "failed"
|
|
147
|
+
_WARNING: Final = "warning"
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
class ComparisonCheck(ProtocolModel):
|
|
151
|
+
"""One named question about a comparison, and its answer.
|
|
152
|
+
|
|
153
|
+
The same shape as
|
|
154
|
+
:class:`~techtree.models.validation.ValidationCheck` and
|
|
155
|
+
:class:`~techtree.verifiers.models.ExecutionCheck`, and for the same
|
|
156
|
+
reason: a reader wants an ordered list of named verdicts, not an exception
|
|
157
|
+
that stops at whichever rule happened to fail first.
|
|
158
|
+
"""
|
|
159
|
+
|
|
160
|
+
id: NonEmptyString
|
|
161
|
+
status: Literal["passed", "failed", "warning"]
|
|
162
|
+
detail: NonEmptyString
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
class ScheduleObservation(ProtocolModel):
|
|
166
|
+
"""How the two variants were placed in time. Spec section 7.9.
|
|
167
|
+
|
|
168
|
+
Operational metadata, deliberately kept apart from the scientific checks: a
|
|
169
|
+
wide launch skew does not invalidate a comparison, it just means the two
|
|
170
|
+
sides saw the provider at more distant moments, and a reader deciding how
|
|
171
|
+
much that matters needs the number rather than a verdict about it.
|
|
172
|
+
"""
|
|
173
|
+
|
|
174
|
+
schedule: VariantSchedule
|
|
175
|
+
start_skew_seconds: float = Field(ge=0.0)
|
|
176
|
+
completion_window_seconds: float = Field(ge=0.0)
|
|
177
|
+
overlapped: bool
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
class PairedReceiptRow(ProtocolModel):
|
|
181
|
+
"""One committed task, and the two receipts that scored it."""
|
|
182
|
+
|
|
183
|
+
position: int = Field(ge=0)
|
|
184
|
+
task_hash: Digest
|
|
185
|
+
baseline_receipt_id: NonEmptyString
|
|
186
|
+
baseline_receipt_digest: Digest
|
|
187
|
+
candidate_receipt_id: NonEmptyString
|
|
188
|
+
candidate_receipt_digest: Digest
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
class RealComparisonResult(ProtocolModel):
|
|
192
|
+
"""Whether the two executions were one experiment, and how it was checked.
|
|
193
|
+
|
|
194
|
+
A local object, versioned by the run directory it lives in rather than by
|
|
195
|
+
the protocol: the frozen v0.1 schema has no controlled-comparison document,
|
|
196
|
+
and inventing one here would be a protocol amendment nobody ratified.
|
|
197
|
+
"""
|
|
198
|
+
|
|
199
|
+
status: ComparisonStatus
|
|
200
|
+
mutation_kind: MutationKind
|
|
201
|
+
manifest_comparison: ManifestComparison
|
|
202
|
+
ordered_task_hashes: list[Digest]
|
|
203
|
+
rows: list[PairedReceiptRow]
|
|
204
|
+
checks: list[ComparisonCheck]
|
|
205
|
+
schedule: ScheduleObservation
|
|
206
|
+
baseline_observed: ObservedSubjectConfiguration
|
|
207
|
+
candidate_observed: ObservedSubjectConfiguration
|
|
208
|
+
|
|
209
|
+
@property
|
|
210
|
+
def failures(self) -> list[ComparisonCheck]:
|
|
211
|
+
"""Return every check that found a scientific invariant broken."""
|
|
212
|
+
return [check for check in self.checks if check.status == _FAILED]
|
|
213
|
+
|
|
214
|
+
@property
|
|
215
|
+
def warnings(self) -> list[ComparisonCheck]:
|
|
216
|
+
"""Return every check that found a weaker claim than it wanted."""
|
|
217
|
+
return [check for check in self.checks if check.status == _WARNING]
|
|
218
|
+
|
|
219
|
+
@property
|
|
220
|
+
def controlled(self) -> bool:
|
|
221
|
+
"""Whether this comparison may carry a scientific result."""
|
|
222
|
+
return self.status in (
|
|
223
|
+
ComparisonStatus.CONTROLLED,
|
|
224
|
+
ComparisonStatus.CONTROLLED_WITH_WARNINGS,
|
|
225
|
+
)
|
|
226
|
+
|
|
227
|
+
@model_validator(mode="after")
|
|
228
|
+
def _check_the_status_is_the_one_the_checks_support(self) -> Self:
|
|
229
|
+
"""Reject a result whose headline disagrees with its own checks."""
|
|
230
|
+
expected = _status_for(self.checks)
|
|
231
|
+
if self.status is not expected:
|
|
232
|
+
raise ValueError(
|
|
233
|
+
f"a comparison whose checks are {expected.value} cannot report "
|
|
234
|
+
f"{self.status.value}"
|
|
235
|
+
)
|
|
236
|
+
if self.status is not ComparisonStatus.INVALID and not self.rows:
|
|
237
|
+
raise ValueError("a controlled comparison pairs at least one task")
|
|
238
|
+
return self
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
@dataclass(frozen=True)
|
|
242
|
+
class ObservedVariant:
|
|
243
|
+
"""Everything about one variant that was measured rather than declared.
|
|
244
|
+
|
|
245
|
+
:class:`~techtree.receipts.observed.ObservedSubjectConfiguration` is the
|
|
246
|
+
fingerprint, and it commits the tool inventory to a single digest. A
|
|
247
|
+
controlled comparison has to look inside that digest — one tool description
|
|
248
|
+
is allowed to differ — so the tools travel beside it, together with the
|
|
249
|
+
effective sampling table the fingerprint also summarizes, the tasks this
|
|
250
|
+
variant actually scored, and the operational envelope its child recorded.
|
|
251
|
+
"""
|
|
252
|
+
|
|
253
|
+
variant: ExperimentVariant
|
|
254
|
+
configuration: ObservedSubjectConfiguration
|
|
255
|
+
tools: list[NormalizedTool]
|
|
256
|
+
sampling: dict[str, JsonValue]
|
|
257
|
+
ordered_task_hashes: list[Digest]
|
|
258
|
+
episode_count: int
|
|
259
|
+
child_outcome: ChildProcessOutcome
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def observe_variant(
|
|
263
|
+
*,
|
|
264
|
+
result: VariantExecutionResult,
|
|
265
|
+
resolved_config: Mapping[str, Any],
|
|
266
|
+
runtime: RuntimeSpec,
|
|
267
|
+
) -> ObservedVariant:
|
|
268
|
+
"""Fingerprint one executed variant from its own evidence.
|
|
269
|
+
|
|
270
|
+
Raises whatever :func:`~techtree.receipts.observed.observed_from_episodes`
|
|
271
|
+
raises: a variant whose own rollouts disagree about what they ran has no
|
|
272
|
+
single observed configuration, and there is nothing to compare.
|
|
273
|
+
"""
|
|
274
|
+
configuration = observed_from_episodes(
|
|
275
|
+
result.episodes,
|
|
276
|
+
resolved_config=resolved_config,
|
|
277
|
+
image_resolution=result.image_resolution,
|
|
278
|
+
runtime=runtime,
|
|
279
|
+
)
|
|
280
|
+
# The same reference rollout the fingerprint was taken from, chosen the
|
|
281
|
+
# same way: every rollout of one variant has already been required to agree
|
|
282
|
+
# with every other, so the first one describes all of them.
|
|
283
|
+
reference = next(trace for episode in result.episodes for trace in episode.traces)
|
|
284
|
+
return ObservedVariant(
|
|
285
|
+
variant=ExperimentVariant(result.variant.value),
|
|
286
|
+
configuration=configuration,
|
|
287
|
+
tools=sorted(reference.tools, key=lambda tool: tool.name),
|
|
288
|
+
sampling=dict(reference.sampling),
|
|
289
|
+
ordered_task_hashes=[
|
|
290
|
+
validate_digest(episode.task_hash) for episode in result.episodes
|
|
291
|
+
],
|
|
292
|
+
episode_count=len(result.episodes),
|
|
293
|
+
child_outcome=result.child_outcome,
|
|
294
|
+
)
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def compare_real_variants(
|
|
298
|
+
*,
|
|
299
|
+
campaign: CampaignSpec,
|
|
300
|
+
baseline_manifest: ExperimentManifest,
|
|
301
|
+
candidate_manifest: ExperimentManifest,
|
|
302
|
+
prepared_manifest_comparison: ManifestComparison,
|
|
303
|
+
baseline_receipts: Sequence[EpisodeReceipt],
|
|
304
|
+
candidate_receipts: Sequence[EpisodeReceipt],
|
|
305
|
+
taskset_lock: TasksetLock,
|
|
306
|
+
baseline_observed: ObservedVariant,
|
|
307
|
+
candidate_observed: ObservedVariant,
|
|
308
|
+
schedule: VariantSchedule,
|
|
309
|
+
) -> RealComparisonResult:
|
|
310
|
+
"""Verify declared and observed control and return the paired rows."""
|
|
311
|
+
committed = [validate_digest(value) for value in taskset_lock.ordered_task_hashes]
|
|
312
|
+
recomputed = compare_manifests(
|
|
313
|
+
baseline_manifest, candidate_manifest, campaign.mutation_contract
|
|
314
|
+
)
|
|
315
|
+
|
|
316
|
+
checks: list[ComparisonCheck] = [
|
|
317
|
+
*_declared_checks(
|
|
318
|
+
campaign=campaign,
|
|
319
|
+
baseline=baseline_manifest,
|
|
320
|
+
candidate=candidate_manifest,
|
|
321
|
+
prepared=prepared_manifest_comparison,
|
|
322
|
+
recomputed=recomputed,
|
|
323
|
+
taskset_lock=taskset_lock,
|
|
324
|
+
),
|
|
325
|
+
*_observed_checks(
|
|
326
|
+
campaign=campaign,
|
|
327
|
+
baseline_manifest=baseline_manifest,
|
|
328
|
+
candidate_manifest=candidate_manifest,
|
|
329
|
+
baseline=baseline_observed,
|
|
330
|
+
candidate=candidate_observed,
|
|
331
|
+
committed=committed,
|
|
332
|
+
),
|
|
333
|
+
]
|
|
334
|
+
rows, pairing = _pair_receipts(
|
|
335
|
+
baseline_receipts=baseline_receipts,
|
|
336
|
+
candidate_receipts=candidate_receipts,
|
|
337
|
+
committed=committed,
|
|
338
|
+
)
|
|
339
|
+
checks.append(pairing)
|
|
340
|
+
|
|
341
|
+
observation = _observe_schedule(
|
|
342
|
+
schedule=schedule,
|
|
343
|
+
baseline=baseline_observed.child_outcome,
|
|
344
|
+
candidate=candidate_observed.child_outcome,
|
|
345
|
+
)
|
|
346
|
+
checks.append(
|
|
347
|
+
_check(
|
|
348
|
+
"schedule_recorded",
|
|
349
|
+
_PASSED,
|
|
350
|
+
f"the variants ran under {schedule.value}, launched "
|
|
351
|
+
f"{observation.start_skew_seconds:.3f}s apart and completed within "
|
|
352
|
+
f"{observation.completion_window_seconds:.3f}s",
|
|
353
|
+
)
|
|
354
|
+
)
|
|
355
|
+
|
|
356
|
+
status = _status_for(checks)
|
|
357
|
+
return RealComparisonResult(
|
|
358
|
+
status=status,
|
|
359
|
+
mutation_kind=campaign.mutation_contract.kind,
|
|
360
|
+
manifest_comparison=recomputed,
|
|
361
|
+
ordered_task_hashes=committed,
|
|
362
|
+
rows=[] if status is ComparisonStatus.INVALID else rows,
|
|
363
|
+
checks=checks,
|
|
364
|
+
schedule=observation,
|
|
365
|
+
baseline_observed=baseline_observed.configuration,
|
|
366
|
+
candidate_observed=candidate_observed.configuration,
|
|
367
|
+
)
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
# ---------------------------------------------------------------------------
|
|
371
|
+
# What the two documents declared
|
|
372
|
+
# ---------------------------------------------------------------------------
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
def _declared_checks(
|
|
376
|
+
*,
|
|
377
|
+
campaign: CampaignSpec,
|
|
378
|
+
baseline: ExperimentManifest,
|
|
379
|
+
candidate: ExperimentManifest,
|
|
380
|
+
prepared: ManifestComparison,
|
|
381
|
+
recomputed: ManifestComparison,
|
|
382
|
+
taskset_lock: TasksetLock,
|
|
383
|
+
) -> list[ComparisonCheck]:
|
|
384
|
+
"""Check every field spec section 7.9 requires the two variants to share.
|
|
385
|
+
|
|
386
|
+
The whole-configuration diff below already covers most of these, and they
|
|
387
|
+
are stated separately anyway. A diff can say that two documents disagree at
|
|
388
|
+
a pointer; it cannot say which disagreement was structurally impossible,
|
|
389
|
+
and an operator reading a failed comparison wants to be told "the two sides
|
|
390
|
+
name a different model", not "``/agents/subject/model/model_id``".
|
|
391
|
+
"""
|
|
392
|
+
left = baseline.configuration
|
|
393
|
+
right = candidate.configuration
|
|
394
|
+
subject_left = _subject(baseline)
|
|
395
|
+
subject_right = _subject(candidate)
|
|
396
|
+
|
|
397
|
+
checks = [
|
|
398
|
+
_same(
|
|
399
|
+
"declared_campaign",
|
|
400
|
+
"Campaign",
|
|
401
|
+
baseline.campaign_spec_digest,
|
|
402
|
+
candidate.campaign_spec_digest,
|
|
403
|
+
),
|
|
404
|
+
_same(
|
|
405
|
+
"declared_data_policy",
|
|
406
|
+
"DataPolicy",
|
|
407
|
+
left.data_policy_digest,
|
|
408
|
+
right.data_policy_digest,
|
|
409
|
+
),
|
|
410
|
+
_same(
|
|
411
|
+
"declared_program_and_context",
|
|
412
|
+
"improvement program or public context",
|
|
413
|
+
(baseline.program_ref, baseline.public_context),
|
|
414
|
+
(candidate.program_ref, candidate.public_context),
|
|
415
|
+
),
|
|
416
|
+
_same(
|
|
417
|
+
"declared_outcome_contract",
|
|
418
|
+
"OutcomeContract",
|
|
419
|
+
left.outcome_contract_digest,
|
|
420
|
+
right.outcome_contract_digest,
|
|
421
|
+
),
|
|
422
|
+
_same(
|
|
423
|
+
"declared_evaluation_backend",
|
|
424
|
+
"evaluation backend",
|
|
425
|
+
left.evaluation_backend,
|
|
426
|
+
right.evaluation_backend,
|
|
427
|
+
),
|
|
428
|
+
_same(
|
|
429
|
+
"declared_environment", "environment", left.environment, right.environment
|
|
430
|
+
),
|
|
431
|
+
_same(
|
|
432
|
+
"declared_model", "subject model", subject_left.model, subject_right.model
|
|
433
|
+
),
|
|
434
|
+
_same(
|
|
435
|
+
"declared_sampling",
|
|
436
|
+
"sampling",
|
|
437
|
+
subject_left.sampling,
|
|
438
|
+
subject_right.sampling,
|
|
439
|
+
),
|
|
440
|
+
_same(
|
|
441
|
+
"declared_harness",
|
|
442
|
+
"harness identity, version or bundled-Skill setting",
|
|
443
|
+
_harness_identity(subject_left),
|
|
444
|
+
_harness_identity(subject_right),
|
|
445
|
+
),
|
|
446
|
+
_same(
|
|
447
|
+
"declared_runtime",
|
|
448
|
+
"subject runtime or image",
|
|
449
|
+
subject_left.runtime,
|
|
450
|
+
subject_right.runtime,
|
|
451
|
+
),
|
|
452
|
+
_same(
|
|
453
|
+
"declared_execution_contract",
|
|
454
|
+
"execution, scoring, evidence or budget contract",
|
|
455
|
+
(left.execution, left.scoring, left.evidence, left.budgets),
|
|
456
|
+
(right.execution, right.scoring, right.evidence, right.budgets),
|
|
457
|
+
),
|
|
458
|
+
*_taskset_checks(campaign, baseline, candidate, taskset_lock),
|
|
459
|
+
*_manifest_comparison_checks(prepared, recomputed),
|
|
460
|
+
_mutation_check(campaign, subject_left, subject_right),
|
|
461
|
+
]
|
|
462
|
+
return checks
|
|
463
|
+
|
|
464
|
+
|
|
465
|
+
def _taskset_checks(
|
|
466
|
+
campaign: CampaignSpec,
|
|
467
|
+
baseline: ExperimentManifest,
|
|
468
|
+
candidate: ExperimentManifest,
|
|
469
|
+
lock: TasksetLock,
|
|
470
|
+
) -> list[ComparisonCheck]:
|
|
471
|
+
"""Require one taskset, one committed membership, and one lock over it."""
|
|
472
|
+
committed = list(campaign.taskset.membership.ordered_task_hashes)
|
|
473
|
+
agreeing = (
|
|
474
|
+
baseline.configuration.taskset == campaign.taskset
|
|
475
|
+
and candidate.configuration.taskset == campaign.taskset
|
|
476
|
+
)
|
|
477
|
+
locked = (
|
|
478
|
+
list(lock.ordered_task_hashes) == committed
|
|
479
|
+
and lock.membership_digest == campaign.taskset.membership.membership_digest
|
|
480
|
+
and lock.membership_digest == membership_digest(committed)
|
|
481
|
+
and lock.taskset_ref == campaign.taskset.ref
|
|
482
|
+
and lock.task_count == len(committed)
|
|
483
|
+
)
|
|
484
|
+
return [
|
|
485
|
+
_check(
|
|
486
|
+
"declared_taskset",
|
|
487
|
+
_PASSED if agreeing else _FAILED,
|
|
488
|
+
(
|
|
489
|
+
"both variants commit to the Campaign's taskset and membership"
|
|
490
|
+
if agreeing
|
|
491
|
+
else "a variant commits to a different taskset or membership "
|
|
492
|
+
"than the Campaign"
|
|
493
|
+
),
|
|
494
|
+
),
|
|
495
|
+
_check(
|
|
496
|
+
"declared_taskset_lock",
|
|
497
|
+
_PASSED if locked else _FAILED,
|
|
498
|
+
(
|
|
499
|
+
f"the lock pins the {len(committed)} tasks the Campaign commits to"
|
|
500
|
+
if locked
|
|
501
|
+
else "the taskset lock does not pin the tasks, the membership "
|
|
502
|
+
"digest or the taskset the Campaign commits to"
|
|
503
|
+
),
|
|
504
|
+
),
|
|
505
|
+
]
|
|
506
|
+
|
|
507
|
+
|
|
508
|
+
def _manifest_comparison_checks(
|
|
509
|
+
prepared: ManifestComparison, recomputed: ManifestComparison
|
|
510
|
+
) -> list[ComparisonCheck]:
|
|
511
|
+
"""Require the recomputed diff to be controlled and to be the prepared one."""
|
|
512
|
+
return [
|
|
513
|
+
_check(
|
|
514
|
+
"declared_only_skill_differs",
|
|
515
|
+
_PASSED if recomputed.controlled else _FAILED,
|
|
516
|
+
(
|
|
517
|
+
"the candidate configuration differs from the baseline only "
|
|
518
|
+
"where the mutation contract permits"
|
|
519
|
+
if recomputed.controlled
|
|
520
|
+
else "; ".join(recomputed.violations)
|
|
521
|
+
),
|
|
522
|
+
),
|
|
523
|
+
_check(
|
|
524
|
+
"declared_comparison_unchanged",
|
|
525
|
+
_PASSED if recomputed == prepared else _FAILED,
|
|
526
|
+
(
|
|
527
|
+
"the comparison recomputed from the executed manifests is the "
|
|
528
|
+
"one the run was prepared with"
|
|
529
|
+
if recomputed == prepared
|
|
530
|
+
else "the executed manifests do not produce the comparison this "
|
|
531
|
+
"run was prepared with"
|
|
532
|
+
),
|
|
533
|
+
),
|
|
534
|
+
]
|
|
535
|
+
|
|
536
|
+
|
|
537
|
+
def _mutation_check(
|
|
538
|
+
campaign: CampaignSpec, baseline: AgentSpec, candidate: AgentSpec
|
|
539
|
+
) -> ComparisonCheck:
|
|
540
|
+
"""Hold the declared skill lists to the shape the mutation kind requires."""
|
|
541
|
+
kind = campaign.mutation_contract.kind
|
|
542
|
+
left = [reference.digest for reference in baseline.harness.skills]
|
|
543
|
+
right = [reference.digest for reference in candidate.harness.skills]
|
|
544
|
+
|
|
545
|
+
if kind is MutationKind.SKILL_INSERTION:
|
|
546
|
+
ok = not left and len(right) == 1
|
|
547
|
+
detail = (
|
|
548
|
+
"the baseline declares no Skill and the candidate declares one"
|
|
549
|
+
if ok
|
|
550
|
+
else f"a skill_insertion declares 0 then 1 Skill; got {len(left)} "
|
|
551
|
+
f"then {len(right)}"
|
|
552
|
+
)
|
|
553
|
+
else:
|
|
554
|
+
ok = len(left) == 1 and len(right) == 1 and left != right
|
|
555
|
+
detail = (
|
|
556
|
+
"the baseline and the candidate each declare one Skill, and they "
|
|
557
|
+
"are different Skills"
|
|
558
|
+
if ok
|
|
559
|
+
else "a skill_replacement declares one differing Skill on each side"
|
|
560
|
+
)
|
|
561
|
+
return _check(f"declared_mutation_{kind.value}", _PASSED if ok else _FAILED, detail)
|
|
562
|
+
|
|
563
|
+
|
|
564
|
+
# ---------------------------------------------------------------------------
|
|
565
|
+
# What the two executions did
|
|
566
|
+
# ---------------------------------------------------------------------------
|
|
567
|
+
|
|
568
|
+
|
|
569
|
+
def _observed_checks(
|
|
570
|
+
*,
|
|
571
|
+
campaign: CampaignSpec,
|
|
572
|
+
baseline_manifest: ExperimentManifest,
|
|
573
|
+
candidate_manifest: ExperimentManifest,
|
|
574
|
+
baseline: ObservedVariant,
|
|
575
|
+
candidate: ObservedVariant,
|
|
576
|
+
committed: Sequence[Digest],
|
|
577
|
+
) -> list[ComparisonCheck]:
|
|
578
|
+
"""Check what actually ran, against the other side and against the manifest."""
|
|
579
|
+
left = baseline.configuration
|
|
580
|
+
right = candidate.configuration
|
|
581
|
+
|
|
582
|
+
return [
|
|
583
|
+
_check(
|
|
584
|
+
"observed_task_order",
|
|
585
|
+
(
|
|
586
|
+
_PASSED
|
|
587
|
+
if baseline.ordered_task_hashes
|
|
588
|
+
== candidate.ordered_task_hashes
|
|
589
|
+
== list(committed)
|
|
590
|
+
else _FAILED
|
|
591
|
+
),
|
|
592
|
+
(
|
|
593
|
+
"both variants scored the committed tasks in committed order"
|
|
594
|
+
if baseline.ordered_task_hashes
|
|
595
|
+
== candidate.ordered_task_hashes
|
|
596
|
+
== list(committed)
|
|
597
|
+
else "a variant scored a different set of tasks, or scored them "
|
|
598
|
+
"in an order the Campaign did not commit to"
|
|
599
|
+
),
|
|
600
|
+
),
|
|
601
|
+
_check(
|
|
602
|
+
"observed_episode_count",
|
|
603
|
+
(
|
|
604
|
+
_PASSED
|
|
605
|
+
if baseline.episode_count == candidate.episode_count == len(committed)
|
|
606
|
+
else _FAILED
|
|
607
|
+
),
|
|
608
|
+
(
|
|
609
|
+
f"both variants recorded {len(committed)} episodes"
|
|
610
|
+
if baseline.episode_count == candidate.episode_count == len(committed)
|
|
611
|
+
else f"the variants recorded {baseline.episode_count} and "
|
|
612
|
+
f"{candidate.episode_count} episodes for {len(committed)} tasks"
|
|
613
|
+
),
|
|
614
|
+
),
|
|
615
|
+
_same("observed_model", "model", left.model_id, right.model_id),
|
|
616
|
+
_same(
|
|
617
|
+
"observed_sampling",
|
|
618
|
+
"effective sampling",
|
|
619
|
+
left.sampling_digest,
|
|
620
|
+
right.sampling_digest,
|
|
621
|
+
),
|
|
622
|
+
_same(
|
|
623
|
+
"observed_harness",
|
|
624
|
+
"harness",
|
|
625
|
+
(left.harness_id, left.harness_version),
|
|
626
|
+
(right.harness_id, right.harness_version),
|
|
627
|
+
),
|
|
628
|
+
_same(
|
|
629
|
+
"observed_bundled_skill",
|
|
630
|
+
"bundled-Skill setting",
|
|
631
|
+
left.use_bundled_skill,
|
|
632
|
+
right.use_bundled_skill,
|
|
633
|
+
),
|
|
634
|
+
_same(
|
|
635
|
+
"observed_runtime_image",
|
|
636
|
+
"subject runtime or image",
|
|
637
|
+
(left.runtime_kind, left.runtime_image, left.runtime_image_index_digest),
|
|
638
|
+
(right.runtime_kind, right.runtime_image, right.runtime_image_index_digest),
|
|
639
|
+
),
|
|
640
|
+
_same(
|
|
641
|
+
"observed_runtime_platform_digest",
|
|
642
|
+
"resolved platform-specific image digest",
|
|
643
|
+
(left.runtime_platform, left.runtime_image_platform_digest),
|
|
644
|
+
(right.runtime_platform, right.runtime_image_platform_digest),
|
|
645
|
+
),
|
|
646
|
+
*_runtime_pin_checks(campaign, baseline, candidate),
|
|
647
|
+
_tool_surface_check(baseline, candidate),
|
|
648
|
+
_same(
|
|
649
|
+
"observed_reward_contract",
|
|
650
|
+
"reward names or weights",
|
|
651
|
+
left.reward_contract_digest,
|
|
652
|
+
right.reward_contract_digest,
|
|
653
|
+
),
|
|
654
|
+
_same(
|
|
655
|
+
"observed_verifiers_build",
|
|
656
|
+
"Verifiers build",
|
|
657
|
+
(left.verifiers_version, left.verifiers_revision),
|
|
658
|
+
(right.verifiers_version, right.verifiers_revision),
|
|
659
|
+
),
|
|
660
|
+
_declared_to_observed(baseline_manifest, baseline),
|
|
661
|
+
_declared_to_observed(candidate_manifest, candidate),
|
|
662
|
+
*weaker_claim_warnings(campaign),
|
|
663
|
+
]
|
|
664
|
+
|
|
665
|
+
|
|
666
|
+
def _tool_surface_check(
|
|
667
|
+
baseline: ObservedVariant, candidate: ObservedVariant
|
|
668
|
+
) -> ComparisonCheck:
|
|
669
|
+
"""Permit the Skill index delta and nothing else. See :data:`SKILL_INDEX_TOOL`."""
|
|
670
|
+
left = {tool.name: tool for tool in baseline.tools}
|
|
671
|
+
right = {tool.name: tool for tool in candidate.tools}
|
|
672
|
+
|
|
673
|
+
departure = _conformance_departure(baseline, candidate)
|
|
674
|
+
if departure is not None:
|
|
675
|
+
return _check("observed_tool_inventory", _FAILED, departure)
|
|
676
|
+
|
|
677
|
+
if set(left) != set(right):
|
|
678
|
+
added = sorted(set(right) - set(left))
|
|
679
|
+
removed = sorted(set(left) - set(right))
|
|
680
|
+
return _check(
|
|
681
|
+
"observed_tool_inventory",
|
|
682
|
+
_FAILED,
|
|
683
|
+
"the two variants were offered different tools "
|
|
684
|
+
f"(added {added or 'nothing'}, removed {removed or 'nothing'})",
|
|
685
|
+
)
|
|
686
|
+
|
|
687
|
+
schemas = sorted(
|
|
688
|
+
name
|
|
689
|
+
for name in left
|
|
690
|
+
if left[name].parameters_digest != right[name].parameters_digest
|
|
691
|
+
)
|
|
692
|
+
if schemas:
|
|
693
|
+
return _check(
|
|
694
|
+
"observed_tool_inventory",
|
|
695
|
+
_FAILED,
|
|
696
|
+
f"the parameter schema of {', '.join(schemas)} differs between the "
|
|
697
|
+
"two variants",
|
|
698
|
+
)
|
|
699
|
+
|
|
700
|
+
descriptions = sorted(
|
|
701
|
+
name
|
|
702
|
+
for name in left
|
|
703
|
+
if left[name].description_digest != right[name].description_digest
|
|
704
|
+
)
|
|
705
|
+
if not descriptions:
|
|
706
|
+
return _check(
|
|
707
|
+
"observed_tool_inventory",
|
|
708
|
+
_PASSED,
|
|
709
|
+
f"both variants were offered the same {len(left)} tools, described "
|
|
710
|
+
"identically",
|
|
711
|
+
)
|
|
712
|
+
if descriptions == [SKILL_INDEX_TOOL]:
|
|
713
|
+
return _check(
|
|
714
|
+
"observed_tool_inventory",
|
|
715
|
+
_PASSED,
|
|
716
|
+
f"both variants were offered the same {len(left)} tools with the "
|
|
717
|
+
f"same schemas; only {SKILL_INDEX_TOOL}'s description differs, "
|
|
718
|
+
"which is where the harness lists the Skills the subject can see",
|
|
719
|
+
)
|
|
720
|
+
return _check(
|
|
721
|
+
"observed_tool_inventory",
|
|
722
|
+
_FAILED,
|
|
723
|
+
f"the description of {', '.join(descriptions)} differs between the two "
|
|
724
|
+
f"variants; only {SKILL_INDEX_TOOL}'s may",
|
|
725
|
+
)
|
|
726
|
+
|
|
727
|
+
|
|
728
|
+
def _conformance_departure(
|
|
729
|
+
baseline: ObservedVariant, candidate: ObservedVariant
|
|
730
|
+
) -> str | None:
|
|
731
|
+
"""Return why the offered tools are not the pinned harness's, or nothing.
|
|
732
|
+
|
|
733
|
+
Decisions document 0007 R9 item 4. A comparison sees one description
|
|
734
|
+
differing on one tool and cannot tell, from inside itself, whether that is
|
|
735
|
+
the Skill index doing its job or a harness that changed underneath the
|
|
736
|
+
Campaign. The pinned fixture is the outside evidence: it fixes the tool
|
|
737
|
+
count, the names and the parameter schemas of the harness build the
|
|
738
|
+
Campaign declares, so anything else is a departure and not a derived
|
|
739
|
+
difference.
|
|
740
|
+
"""
|
|
741
|
+
for observed in (baseline, candidate):
|
|
742
|
+
configuration = observed.configuration
|
|
743
|
+
try:
|
|
744
|
+
pinned = harness_conformance(
|
|
745
|
+
configuration.harness_id, configuration.harness_version
|
|
746
|
+
)
|
|
747
|
+
except FileNotFoundError:
|
|
748
|
+
return (
|
|
749
|
+
f"no tool surface was ever recorded for "
|
|
750
|
+
f"{configuration.harness_id} {configuration.harness_version}, "
|
|
751
|
+
"so the difference between the two variants cannot be shown to "
|
|
752
|
+
"be the Skill index alone"
|
|
753
|
+
)
|
|
754
|
+
offered = sorted(tool.name for tool in observed.tools)
|
|
755
|
+
if offered != sorted(pinned.tool_names):
|
|
756
|
+
return (
|
|
757
|
+
f"the {observed.variant.value} was offered {len(offered)} tools "
|
|
758
|
+
f"where {configuration.harness_id} "
|
|
759
|
+
f"{configuration.harness_version} offers "
|
|
760
|
+
f"{len(pinned.tool_names)}, so the harness is not the one the "
|
|
761
|
+
"Campaign declares"
|
|
762
|
+
)
|
|
763
|
+
expected = pinned.parameters_by_tool
|
|
764
|
+
reshaped = sorted(
|
|
765
|
+
tool.name
|
|
766
|
+
for tool in observed.tools
|
|
767
|
+
if tool.parameters_digest != expected[tool.name]
|
|
768
|
+
)
|
|
769
|
+
if reshaped:
|
|
770
|
+
return (
|
|
771
|
+
f"the {observed.variant.value} was offered a different "
|
|
772
|
+
f"parameter schema for {', '.join(reshaped)} than "
|
|
773
|
+
f"{configuration.harness_id} {configuration.harness_version} "
|
|
774
|
+
"records"
|
|
775
|
+
)
|
|
776
|
+
if pinned.skill_index_tool != SKILL_INDEX_TOOL:
|
|
777
|
+
return (
|
|
778
|
+
f"{configuration.harness_id} {configuration.harness_version} "
|
|
779
|
+
f"renders its Skill index into {pinned.skill_index_tool}, not "
|
|
780
|
+
f"{SKILL_INDEX_TOOL}"
|
|
781
|
+
)
|
|
782
|
+
return None
|
|
783
|
+
|
|
784
|
+
|
|
785
|
+
def _declared_to_observed(
|
|
786
|
+
manifest: ExperimentManifest, observed: ObservedVariant
|
|
787
|
+
) -> ComparisonCheck:
|
|
788
|
+
"""Check one variant's execution against the manifest it was supposed to be.
|
|
789
|
+
|
|
790
|
+
Spec section 7.8 names this ``compare_declared_to_observed``. It lives here
|
|
791
|
+
because a check that only has meaning inside a controlled comparison should
|
|
792
|
+
be read beside the other checks of that comparison, and because
|
|
793
|
+
``ComparisonCheck`` is this section's type.
|
|
794
|
+
"""
|
|
795
|
+
subject = _subject(manifest)
|
|
796
|
+
configuration = observed.configuration
|
|
797
|
+
declared_sampling: dict[str, JsonValue] = {
|
|
798
|
+
"max_tokens": subject.sampling.max_tokens,
|
|
799
|
+
"temperature": subject.sampling.temperature,
|
|
800
|
+
}
|
|
801
|
+
mismatches = [
|
|
802
|
+
label
|
|
803
|
+
for label, declared, seen in (
|
|
804
|
+
("model", subject.model.model_id, configuration.model_id),
|
|
805
|
+
("harness", subject.harness.id, configuration.harness_id),
|
|
806
|
+
("harness version", subject.harness.version, configuration.harness_version),
|
|
807
|
+
(
|
|
808
|
+
"bundled-Skill setting",
|
|
809
|
+
subject.harness.use_bundled_skill,
|
|
810
|
+
configuration.use_bundled_skill,
|
|
811
|
+
),
|
|
812
|
+
("runtime", subject.runtime.type, configuration.runtime_kind),
|
|
813
|
+
("runtime image", subject.runtime.image, configuration.runtime_image),
|
|
814
|
+
(
|
|
815
|
+
"Skill list",
|
|
816
|
+
[reference.digest for reference in subject.harness.skills],
|
|
817
|
+
list(configuration.skill_root_digests),
|
|
818
|
+
),
|
|
819
|
+
# Every parameter the engine resolved, and no parameter it did not:
|
|
820
|
+
# an undeclared sampling key is a difference the manifest never
|
|
821
|
+
# authorized, whichever variant picked it up.
|
|
822
|
+
("sampling", declared_sampling, dict(observed.sampling)),
|
|
823
|
+
)
|
|
824
|
+
if declared != seen
|
|
825
|
+
]
|
|
826
|
+
variant = observed.variant.value
|
|
827
|
+
return _check(
|
|
828
|
+
f"observed_matches_declared_{variant}",
|
|
829
|
+
_PASSED if not mismatches else _FAILED,
|
|
830
|
+
(
|
|
831
|
+
f"the {variant} executed the subject its manifest declares"
|
|
832
|
+
if not mismatches
|
|
833
|
+
else f"the {variant} executed a different {', '.join(mismatches)} "
|
|
834
|
+
"than its manifest declares"
|
|
835
|
+
),
|
|
836
|
+
)
|
|
837
|
+
|
|
838
|
+
|
|
839
|
+
def _runtime_pin_checks(
|
|
840
|
+
campaign: CampaignSpec, baseline: ObservedVariant, candidate: ObservedVariant
|
|
841
|
+
) -> list[ComparisonCheck]:
|
|
842
|
+
"""Hold both executions to the container the Campaign pinned.
|
|
843
|
+
|
|
844
|
+
Decisions document 0007 R5. This used to be a warning, because the only
|
|
845
|
+
evidence of what ran was the reference the Campaign had asked for: the
|
|
846
|
+
pinned Verifiers build records no resolved digest, so "the image is the one
|
|
847
|
+
we asked for" was the strongest true statement available. It is a check now
|
|
848
|
+
because the Campaign pins a manifest digest per platform and every run asks
|
|
849
|
+
the local daemon what it holds, so the claim has evidence under it. An
|
|
850
|
+
execution that cannot be tied to the pin is invalid rather than warned
|
|
851
|
+
about — a container nobody can name is not a weaker result.
|
|
852
|
+
"""
|
|
853
|
+
runtime = campaign.subject.runtime
|
|
854
|
+
pinned = runtime.image_index_digest
|
|
855
|
+
matched = sorted(
|
|
856
|
+
observed.variant.value
|
|
857
|
+
for observed in (baseline, candidate)
|
|
858
|
+
if observed.configuration.runtime_image_index_digest == pinned
|
|
859
|
+
)
|
|
860
|
+
both = len(matched) == 2
|
|
861
|
+
|
|
862
|
+
platforms = sorted(
|
|
863
|
+
{observed.configuration.runtime_platform for observed in (baseline, candidate)}
|
|
864
|
+
)
|
|
865
|
+
pinned_platform = len(platforms) == 1 and platforms[0] in (
|
|
866
|
+
runtime.image_platform_digests
|
|
867
|
+
)
|
|
868
|
+
return [
|
|
869
|
+
_check(
|
|
870
|
+
"observed_runtime_image_pinned",
|
|
871
|
+
_PASSED if both else _FAILED,
|
|
872
|
+
(
|
|
873
|
+
"the daemon confirmed both variants ran the image content the "
|
|
874
|
+
"Campaign pinned"
|
|
875
|
+
if both
|
|
876
|
+
else "the container the daemon holds is not the one the Campaign pinned"
|
|
877
|
+
),
|
|
878
|
+
),
|
|
879
|
+
_check(
|
|
880
|
+
"observed_runtime_platform_pinned",
|
|
881
|
+
_PASSED if pinned_platform else _FAILED,
|
|
882
|
+
(
|
|
883
|
+
f"both variants were served on {platforms[0]}, a platform the "
|
|
884
|
+
"Campaign pins a manifest digest for"
|
|
885
|
+
if pinned_platform
|
|
886
|
+
else "the variants were served on platforms the Campaign does "
|
|
887
|
+
"not pin one manifest digest for"
|
|
888
|
+
),
|
|
889
|
+
),
|
|
890
|
+
]
|
|
891
|
+
|
|
892
|
+
|
|
893
|
+
def weaker_claim_warnings(campaign: CampaignSpec) -> list[ComparisonCheck]:
|
|
894
|
+
"""Record the one fact about a real run that cannot be independently pinned.
|
|
895
|
+
|
|
896
|
+
It is not a scientific failure and it may not be left unsaid. Presenting
|
|
897
|
+
"both variants named the same model" as "both variants ran the same model
|
|
898
|
+
build" would be making the stronger claim from the weaker evidence, which is
|
|
899
|
+
the whole thing a controlled comparison exists to stop. Decisions document
|
|
900
|
+
0007 R5 accepts it for v0.1 and forbids suppressing it to obtain the word
|
|
901
|
+
"controlled".
|
|
902
|
+
"""
|
|
903
|
+
if campaign.subject.model.revision is not None:
|
|
904
|
+
return []
|
|
905
|
+
return [
|
|
906
|
+
_check(
|
|
907
|
+
MODEL_REVISION_UNDISCOVERABLE,
|
|
908
|
+
_WARNING,
|
|
909
|
+
f"the provider publishes no revision for "
|
|
910
|
+
f"{campaign.subject.model.model_id}, so both variants are known "
|
|
911
|
+
"to have used the same model identifier and not the same "
|
|
912
|
+
"model build",
|
|
913
|
+
)
|
|
914
|
+
]
|
|
915
|
+
|
|
916
|
+
|
|
917
|
+
# ---------------------------------------------------------------------------
|
|
918
|
+
# The join
|
|
919
|
+
# ---------------------------------------------------------------------------
|
|
920
|
+
|
|
921
|
+
|
|
922
|
+
def _pair_receipts(
|
|
923
|
+
*,
|
|
924
|
+
baseline_receipts: Sequence[EpisodeReceipt],
|
|
925
|
+
candidate_receipts: Sequence[EpisodeReceipt],
|
|
926
|
+
committed: Sequence[Digest],
|
|
927
|
+
) -> tuple[list[PairedReceiptRow], ComparisonCheck]:
|
|
928
|
+
"""Join the two sides task by task, in committed order.
|
|
929
|
+
|
|
930
|
+
A missing task and a duplicated task are both reported rather than raised,
|
|
931
|
+
because they are findings about the comparison and belong in its list of
|
|
932
|
+
checks beside every other finding.
|
|
933
|
+
"""
|
|
934
|
+
left, left_faults = _by_task(baseline_receipts, committed, "baseline")
|
|
935
|
+
right, right_faults = _by_task(candidate_receipts, committed, "candidate")
|
|
936
|
+
faults = [*left_faults, *right_faults]
|
|
937
|
+
if faults:
|
|
938
|
+
return [], _check("paired_task_rewards", _FAILED, "; ".join(faults))
|
|
939
|
+
|
|
940
|
+
rows = [
|
|
941
|
+
PairedReceiptRow(
|
|
942
|
+
position=position,
|
|
943
|
+
task_hash=task_hash,
|
|
944
|
+
baseline_receipt_id=left[task_hash].id,
|
|
945
|
+
baseline_receipt_digest=digest_object(left[task_hash]),
|
|
946
|
+
candidate_receipt_id=right[task_hash].id,
|
|
947
|
+
candidate_receipt_digest=digest_object(right[task_hash]),
|
|
948
|
+
)
|
|
949
|
+
for position, task_hash in enumerate(committed)
|
|
950
|
+
]
|
|
951
|
+
return rows, _check(
|
|
952
|
+
"paired_task_rewards",
|
|
953
|
+
_PASSED,
|
|
954
|
+
f"every one of the {len(rows)} committed tasks is scored exactly once "
|
|
955
|
+
"on each side",
|
|
956
|
+
)
|
|
957
|
+
|
|
958
|
+
|
|
959
|
+
def _by_task(
|
|
960
|
+
receipts: Sequence[EpisodeReceipt],
|
|
961
|
+
committed: Sequence[Digest],
|
|
962
|
+
label: str,
|
|
963
|
+
) -> tuple[dict[Digest, EpisodeReceipt], list[str]]:
|
|
964
|
+
"""Index one side's receipts by task and report what does not line up."""
|
|
965
|
+
by_task: dict[Digest, EpisodeReceipt] = {}
|
|
966
|
+
faults: list[str] = []
|
|
967
|
+
for receipt in receipts:
|
|
968
|
+
if receipt.task_hash in by_task:
|
|
969
|
+
faults.append(f"the {label} scored {receipt.task_hash} twice")
|
|
970
|
+
continue
|
|
971
|
+
by_task[receipt.task_hash] = receipt
|
|
972
|
+
|
|
973
|
+
missing = [value for value in committed if value not in by_task]
|
|
974
|
+
unexpected = sorted(set(by_task) - set(committed))
|
|
975
|
+
if missing:
|
|
976
|
+
faults.append(f"the {label} did not score {len(missing)} committed task(s)")
|
|
977
|
+
if unexpected:
|
|
978
|
+
faults.append(
|
|
979
|
+
f"the {label} scored {len(unexpected)} task(s) the Campaign does "
|
|
980
|
+
"not commit to"
|
|
981
|
+
)
|
|
982
|
+
return by_task, faults
|
|
983
|
+
|
|
984
|
+
|
|
985
|
+
# ---------------------------------------------------------------------------
|
|
986
|
+
# Placement in time
|
|
987
|
+
# ---------------------------------------------------------------------------
|
|
988
|
+
|
|
989
|
+
|
|
990
|
+
def _observe_schedule(
|
|
991
|
+
*,
|
|
992
|
+
schedule: VariantSchedule,
|
|
993
|
+
baseline: ChildProcessOutcome,
|
|
994
|
+
candidate: ChildProcessOutcome,
|
|
995
|
+
) -> ScheduleObservation:
|
|
996
|
+
"""Record how far apart the two children started and finished."""
|
|
997
|
+
started = (baseline.started_at, candidate.started_at)
|
|
998
|
+
finished = (baseline.finished_at, candidate.finished_at)
|
|
999
|
+
return ScheduleObservation(
|
|
1000
|
+
schedule=schedule,
|
|
1001
|
+
start_skew_seconds=abs((started[1] - started[0]).total_seconds()),
|
|
1002
|
+
completion_window_seconds=(max(finished) - min(started)).total_seconds(),
|
|
1003
|
+
overlapped=max(started) < min(finished),
|
|
1004
|
+
)
|
|
1005
|
+
|
|
1006
|
+
|
|
1007
|
+
# ---------------------------------------------------------------------------
|
|
1008
|
+
# Small shared pieces
|
|
1009
|
+
# ---------------------------------------------------------------------------
|
|
1010
|
+
|
|
1011
|
+
|
|
1012
|
+
def _status_for(checks: Sequence[ComparisonCheck]) -> ComparisonStatus:
|
|
1013
|
+
"""Return the one status a list of checks supports. Spec section 7.9."""
|
|
1014
|
+
if any(check.status == _FAILED for check in checks):
|
|
1015
|
+
return ComparisonStatus.INVALID
|
|
1016
|
+
if any(check.status == _WARNING for check in checks):
|
|
1017
|
+
return ComparisonStatus.CONTROLLED_WITH_WARNINGS
|
|
1018
|
+
return ComparisonStatus.CONTROLLED
|
|
1019
|
+
|
|
1020
|
+
|
|
1021
|
+
def _check(identifier: str, status: str, detail: str) -> ComparisonCheck:
|
|
1022
|
+
"""Build one check, validated as the model it is."""
|
|
1023
|
+
return ComparisonCheck.model_validate(
|
|
1024
|
+
{"id": identifier, "status": status, "detail": detail}
|
|
1025
|
+
)
|
|
1026
|
+
|
|
1027
|
+
|
|
1028
|
+
def _same(identifier: str, label: str, left: object, right: object) -> ComparisonCheck:
|
|
1029
|
+
"""Report whether two sides agree, without printing what they hold.
|
|
1030
|
+
|
|
1031
|
+
The values themselves are deliberately not in the detail. A model
|
|
1032
|
+
identifier is harmless, a resolved configuration is not, and one rule for
|
|
1033
|
+
all of them is the rule that cannot leak.
|
|
1034
|
+
"""
|
|
1035
|
+
agree = left == right
|
|
1036
|
+
return _check(
|
|
1037
|
+
identifier,
|
|
1038
|
+
_PASSED if agree else _FAILED,
|
|
1039
|
+
(
|
|
1040
|
+
f"the baseline and the candidate share one {label}"
|
|
1041
|
+
if agree
|
|
1042
|
+
else f"the baseline and the candidate do not share one {label}"
|
|
1043
|
+
),
|
|
1044
|
+
)
|
|
1045
|
+
|
|
1046
|
+
|
|
1047
|
+
def _subject(manifest: ExperimentManifest) -> AgentSpec:
|
|
1048
|
+
"""Return the manifest's subject agent, which its own validator requires."""
|
|
1049
|
+
subject = manifest.configuration.agents.get(SUBJECT_AGENT)
|
|
1050
|
+
if subject is None: # pragma: no cover - the model refuses to be built without one
|
|
1051
|
+
raise ValueError("an experiment configuration defines a subject agent")
|
|
1052
|
+
return subject
|
|
1053
|
+
|
|
1054
|
+
|
|
1055
|
+
def _harness_identity(subject: AgentSpec) -> tuple[str, str, bool]:
|
|
1056
|
+
"""Return the part of a harness two variants must share.
|
|
1057
|
+
|
|
1058
|
+
Not the skill list: that is the one field the mutation contract permits to
|
|
1059
|
+
differ, and comparing it here would fail every correct comparison.
|
|
1060
|
+
"""
|
|
1061
|
+
return (
|
|
1062
|
+
subject.harness.id,
|
|
1063
|
+
subject.harness.version,
|
|
1064
|
+
subject.harness.use_bundled_skill,
|
|
1065
|
+
)
|