techtree 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- techtree/__init__.py +35 -0
- techtree/__main__.py +14 -0
- techtree/canonical.py +239 -0
- techtree/catalog/__init__.py +25 -0
- techtree/catalog/repository.py +400 -0
- techtree/catalog/service.py +419 -0
- techtree/cli/__init__.py +1 -0
- techtree/cli/app.py +416 -0
- techtree/cli/commands/__init__.py +1 -0
- techtree/cli/commands/climb.py +1223 -0
- techtree/cli/commands/doctor.py +147 -0
- techtree/cli/commands/engine.py +207 -0
- techtree/cli/commands/proof.py +556 -0
- techtree/cli/commands/publish.py +447 -0
- techtree/cli/commands/release.py +303 -0
- techtree/cli/commands/run.py +1067 -0
- techtree/cli/commands/setup.py +181 -0
- techtree/cli/commands/skill.py +221 -0
- techtree/cli/commands/uplift.py +698 -0
- techtree/cli/commands/withdraw.py +212 -0
- techtree/cli/confirm.py +47 -0
- techtree/cli/context.py +96 -0
- techtree/cli/invoke.py +220 -0
- techtree/cli/output.py +280 -0
- techtree/constants.py +138 -0
- techtree/crypto.py +128 -0
- techtree/doctor/__init__.py +1 -0
- techtree/doctor/checks.py +675 -0
- techtree/doctor/execution_checks.py +435 -0
- techtree/doctor/service.py +326 -0
- techtree/drafts/__init__.py +32 -0
- techtree/drafts/source.py +146 -0
- techtree/drafts/store.py +992 -0
- techtree/engines/__init__.py +1 -0
- techtree/engines/bundle.py +251 -0
- techtree/engines/installer.py +679 -0
- techtree/engines/registry.py +235 -0
- techtree/engines/runner.py +170 -0
- techtree/errors.py +262 -0
- techtree/fs.py +234 -0
- techtree/harness.py +108 -0
- techtree/identity/__init__.py +41 -0
- techtree/identity/models.py +113 -0
- techtree/identity/service.py +199 -0
- techtree/identity/store.py +263 -0
- techtree/ids.py +85 -0
- techtree/manifests/__init__.py +39 -0
- techtree/manifests/builder.py +433 -0
- techtree/manifests/compare.py +376 -0
- techtree/models/__init__.py +282 -0
- techtree/models/base.py +201 -0
- techtree/models/campaign.py +484 -0
- techtree/models/catalog.py +227 -0
- techtree/models/cli.py +151 -0
- techtree/models/climb.py +254 -0
- techtree/models/data_policy.py +130 -0
- techtree/models/engine.py +156 -0
- techtree/models/episode_receipt.py +130 -0
- techtree/models/evaluation_backend.py +113 -0
- techtree/models/experiment.py +154 -0
- techtree/models/run.py +214 -0
- techtree/models/skill.py +156 -0
- techtree/models/uplift_report.py +158 -0
- techtree/models/validation.py +299 -0
- techtree/paths.py +116 -0
- techtree/presentation/__init__.py +31 -0
- techtree/presentation/build.py +1242 -0
- techtree/presentation/compact.py +246 -0
- techtree/presentation/evidence.py +169 -0
- techtree/presentation/models.py +358 -0
- techtree/presentation/rich.py +312 -0
- techtree/presentation/sanitize.py +156 -0
- techtree/publication/__init__.py +44 -0
- techtree/publication/address.py +180 -0
- techtree/publication/coordinates.py +26 -0
- techtree/publication/journal.py +212 -0
- techtree/publication/keccak.py +183 -0
- techtree/publication/models.py +209 -0
- techtree/publication/offer.py +35 -0
- techtree/publication/service.py +618 -0
- techtree/publication/transport.py +296 -0
- techtree/publication/verify.py +242 -0
- techtree/publication/withdraw.py +156 -0
- techtree/py.typed +0 -0
- techtree/receipts/__init__.py +52 -0
- techtree/receipts/bundle.py +578 -0
- techtree/receipts/compare.py +1065 -0
- techtree/receipts/episode.py +672 -0
- techtree/receipts/execution.py +630 -0
- techtree/receipts/observed.py +474 -0
- techtree/receipts/set.py +336 -0
- techtree/receipts/uplift.py +655 -0
- techtree/receipts/verify.py +1055 -0
- techtree/release/__init__.py +9 -0
- techtree/release/bootstrap.py +509 -0
- techtree/release/checks.py +376 -0
- techtree/release/document.py +125 -0
- techtree/release/generate.py +221 -0
- techtree/release/models.py +293 -0
- techtree/release/provenance.py +109 -0
- techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
- techtree/resources/catalog/catalog.json +32 -0
- techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
- techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
- techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
- techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
- techtree/resources/engines/default/engine.json +20 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
- techtree/resources/engines/default/pyproject.toml +23 -0
- techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
- techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
- techtree/resources/engines/default/tools/normalize_validation.py +222 -0
- techtree/resources/engines/default/uv.lock +1758 -0
- techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
- techtree/resources/release/build-provenance.json +4 -0
- techtree/resources/release/release-core.json +24 -0
- techtree/runs/__init__.py +31 -0
- techtree/runs/artifacts.py +750 -0
- techtree/runs/child_registry.py +228 -0
- techtree/runs/events.py +478 -0
- techtree/runs/executor.py +140 -0
- techtree/runs/fake.py +741 -0
- techtree/runs/launcher.py +253 -0
- techtree/runs/machine.py +489 -0
- techtree/runs/real.py +789 -0
- techtree/runs/service.py +616 -0
- techtree/runs/store.py +555 -0
- techtree/runs/validation.py +259 -0
- techtree/runs/variants.py +684 -0
- techtree/settings.py +143 -0
- techtree/skills/__init__.py +14 -0
- techtree/skills/archive.py +282 -0
- techtree/skills/policy.py +62 -0
- techtree/skills/scanner.py +394 -0
- techtree/skills/service.py +752 -0
- techtree/skills/starter.py +434 -0
- techtree/tasksets/__init__.py +1 -0
- techtree/tasksets/membership.py +269 -0
- techtree/tasksets/provider.py +207 -0
- techtree/tasksets/resolver.py +311 -0
- techtree/tasksets/service.py +484 -0
- techtree/tasksets/verifiers_cli.py +538 -0
- techtree/uplift/__init__.py +20 -0
- techtree/uplift/context.py +544 -0
- techtree/uplift/derive.py +203 -0
- techtree/uplift/public_tasks.py +151 -0
- techtree/uplift/service.py +719 -0
- techtree/uplift/source.py +160 -0
- techtree/verifiers/__init__.py +31 -0
- techtree/verifiers/budget.py +219 -0
- techtree/verifiers/child.py +633 -0
- techtree/verifiers/compiler.py +432 -0
- techtree/verifiers/config.py +365 -0
- techtree/verifiers/credentials.py +321 -0
- techtree/verifiers/image.py +126 -0
- techtree/verifiers/models.py +527 -0
- techtree/verifiers/outputs.py +368 -0
- techtree/verifiers/progress.py +192 -0
- techtree/verifiers/supervisor.py +341 -0
- techtree/verifiers/verify.py +782 -0
- techtree/version.py +39 -0
- techtree/worker/__init__.py +18 -0
- techtree/worker/execute.py +487 -0
- techtree/worker/main.py +57 -0
- techtree-0.1.0.dist-info/METADATA +344 -0
- techtree-0.1.0.dist-info/RECORD +174 -0
- techtree-0.1.0.dist-info/WHEEL +4 -0
- techtree-0.1.0.dist-info/entry_points.txt +3 -0
- techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,672 @@
|
|
|
1
|
+
"""One receipt per committed task, from recorded evidence. Spec section 7.6.
|
|
2
|
+
|
|
3
|
+
The input is the engine's own normalized projection of one variant's episodes
|
|
4
|
+
and the run's immutable lineage. The output is one
|
|
5
|
+
:class:`~techtree.models.episode_receipt.EpisodeReceipt` per task the Campaign
|
|
6
|
+
committed to, in committed order, carrying the reward Verifiers recorded and
|
|
7
|
+
nothing derived from it.
|
|
8
|
+
|
|
9
|
+
Three boundaries meet in this module and each is drawn deliberately.
|
|
10
|
+
|
|
11
|
+
*The evidence boundary.* :func:`read_variant_episodes` is the only door the
|
|
12
|
+
normalized file comes through, and it maps every way that file can be unusable
|
|
13
|
+
onto the spec section 15 vocabulary: absent evidence is
|
|
14
|
+
``evaluation_output_missing``, unreadable evidence is
|
|
15
|
+
``evaluation_output_corrupt``, and a reward that is not a number is
|
|
16
|
+
``reward_non_finite`` rather than a generic parse failure — a run whose scoring
|
|
17
|
+
produced ``NaN`` failed in a specific way and a reader deserves to be told
|
|
18
|
+
which.
|
|
19
|
+
|
|
20
|
+
*The task-hash boundary.* Verifiers spells a task hash as bare hexadecimal and
|
|
21
|
+
Techtree spells it as a prefixed digest. The engine's normalizer converts it
|
|
22
|
+
once, on the side that read the wire record, and every hash that enters this
|
|
23
|
+
module is revalidated as a Techtree digest before it is compared to anything.
|
|
24
|
+
A type annotation is a claim about the caller; a commitment other parties are
|
|
25
|
+
held to is checked.
|
|
26
|
+
|
|
27
|
+
*The lineage boundary.* A receipt names the Campaign, the improvement program,
|
|
28
|
+
the public context, the DataPolicy, the OutcomeContract, the evaluation backend
|
|
29
|
+
and the experiment manifest it was produced under. Every one of those is copied
|
|
30
|
+
from the run's own immutable request and the run's own staged manifest, and
|
|
31
|
+
every one is required to agree with the others before a receipt is built.
|
|
32
|
+
Nothing is defaulted and nothing is re-resolved.
|
|
33
|
+
|
|
34
|
+
What is *not* a refusal matters as much. An episode that failed, a trace that
|
|
35
|
+
errored, a rollout the provider gave up on: those produce receipts with
|
|
36
|
+
``score_status`` saying so. The refusals are reserved for evidence that cannot
|
|
37
|
+
be joined onto the Campaign's commitment at all, because those are the cases
|
|
38
|
+
where continuing would produce a tidy document making a false claim.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
from __future__ import annotations
|
|
42
|
+
|
|
43
|
+
import json
|
|
44
|
+
import math
|
|
45
|
+
from collections.abc import Sequence
|
|
46
|
+
from pathlib import Path
|
|
47
|
+
from typing import Final
|
|
48
|
+
|
|
49
|
+
from techtree.canonical import (
|
|
50
|
+
canonical_json_bytes,
|
|
51
|
+
digest_object,
|
|
52
|
+
sha256_digest_bytes,
|
|
53
|
+
validate_digest,
|
|
54
|
+
)
|
|
55
|
+
from techtree.constants import EPISODE_RECEIPT_SCHEMA_VERSION
|
|
56
|
+
from techtree.errors import ValidationError, VerificationError
|
|
57
|
+
from techtree.models.base import ArtifactRef, Digest, JsonValue
|
|
58
|
+
from techtree.models.campaign import SUBJECT_AGENT, EvidenceRequirements
|
|
59
|
+
from techtree.models.episode_receipt import (
|
|
60
|
+
EpisodeReceipt,
|
|
61
|
+
EvidenceStatus,
|
|
62
|
+
NamedTraceReceipt,
|
|
63
|
+
ScoreStatus,
|
|
64
|
+
SubjectRuntimeReceipt,
|
|
65
|
+
)
|
|
66
|
+
from techtree.models.evaluation_backend import EvaluationBackendSpec
|
|
67
|
+
from techtree.models.experiment import ExperimentManifest, ExperimentVariant
|
|
68
|
+
from techtree.models.run import RunRequest
|
|
69
|
+
from techtree.verifiers.models import (
|
|
70
|
+
NormalizedEpisode,
|
|
71
|
+
NormalizedTrace,
|
|
72
|
+
VariantExecutionResult,
|
|
73
|
+
VariantName,
|
|
74
|
+
)
|
|
75
|
+
from techtree.verifiers.outputs import read_normalized_episodes
|
|
76
|
+
|
|
77
|
+
__all__ = [
|
|
78
|
+
"EPISODE_COUNT_MISMATCH",
|
|
79
|
+
"EPISODE_RECEIPT_INVALID",
|
|
80
|
+
"EVALUATION_OUTPUT_CORRUPT",
|
|
81
|
+
"EVALUATION_OUTPUT_MISSING",
|
|
82
|
+
"REWARD_MISSING",
|
|
83
|
+
"REWARD_NON_FINITE",
|
|
84
|
+
"TASK_MEMBERSHIP_MISMATCH",
|
|
85
|
+
"TRACE_ROLE_MISMATCH",
|
|
86
|
+
"build_episode_receipt",
|
|
87
|
+
"build_variant_receipts",
|
|
88
|
+
"experiment_variant_of",
|
|
89
|
+
"read_variant_episodes",
|
|
90
|
+
]
|
|
91
|
+
|
|
92
|
+
#: Stable error codes. Spec section 15 fixes the vocabulary; this module is
|
|
93
|
+
#: where the evidence-side half of it is raised.
|
|
94
|
+
EVALUATION_OUTPUT_MISSING: Final = "evaluation_output_missing"
|
|
95
|
+
EVALUATION_OUTPUT_CORRUPT: Final = "evaluation_output_corrupt"
|
|
96
|
+
EPISODE_COUNT_MISMATCH: Final = "episode_count_mismatch"
|
|
97
|
+
TASK_MEMBERSHIP_MISMATCH: Final = "task_membership_mismatch"
|
|
98
|
+
TRACE_ROLE_MISMATCH: Final = "trace_role_mismatch"
|
|
99
|
+
REWARD_MISSING: Final = "reward_missing"
|
|
100
|
+
REWARD_NON_FINITE: Final = "reward_non_finite"
|
|
101
|
+
|
|
102
|
+
#: A receipt whose lineage does not hold together. Spec section 15 lists the
|
|
103
|
+
#: evidence failures; this is the one that says the *provenance* disagrees,
|
|
104
|
+
#: which is a different question and deserves its own code.
|
|
105
|
+
EPISODE_RECEIPT_INVALID: Final = "episode_receipt_invalid"
|
|
106
|
+
|
|
107
|
+
#: How many hexadecimal characters a derived identifier carries, matching the
|
|
108
|
+
#: 32 that :mod:`techtree.ids` issues.
|
|
109
|
+
_DERIVED_ID_LENGTH: Final = 32
|
|
110
|
+
|
|
111
|
+
#: The JSON key that makes a non-finite number under it a reward failure
|
|
112
|
+
#: rather than a generic corruption.
|
|
113
|
+
_REWARD_KEY: Final = "rewards"
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def experiment_variant_of(variant: VariantName) -> ExperimentVariant:
|
|
117
|
+
"""Return the protocol variant one execution-local variant names.
|
|
118
|
+
|
|
119
|
+
Two enumerations spell the same two words: one belongs to the execution
|
|
120
|
+
layer, which schedules children, and one belongs to the protocol, which
|
|
121
|
+
describes manifests and receipts. Converting between them explicitly, in
|
|
122
|
+
one place, keeps the layers separate without letting either drift.
|
|
123
|
+
"""
|
|
124
|
+
return ExperimentVariant(variant.value)
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
# ---------------------------------------------------------------------------
|
|
128
|
+
# The evidence boundary
|
|
129
|
+
# ---------------------------------------------------------------------------
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def read_variant_episodes(path: Path) -> list[NormalizedEpisode]:
|
|
133
|
+
"""Read one variant's normalized episodes, or say precisely what is wrong.
|
|
134
|
+
|
|
135
|
+
The raw ``traces.jsonl`` is deliberately not an accepted input here. It is
|
|
136
|
+
read once, by the pinned Verifiers build that wrote it, inside the engine
|
|
137
|
+
bundle; this is the only projection Techtree interprets.
|
|
138
|
+
"""
|
|
139
|
+
try:
|
|
140
|
+
raw = path.read_bytes()
|
|
141
|
+
except OSError as error:
|
|
142
|
+
raise ValidationError(
|
|
143
|
+
"this variant's normalized evaluation output was not found, so "
|
|
144
|
+
"there is no evidence to build receipts from",
|
|
145
|
+
code=EVALUATION_OUTPUT_MISSING,
|
|
146
|
+
details={"path": str(path)},
|
|
147
|
+
) from error
|
|
148
|
+
|
|
149
|
+
if not raw:
|
|
150
|
+
raise ValidationError(
|
|
151
|
+
"this variant's normalized evaluation output is empty, so no "
|
|
152
|
+
"episode was ever recorded",
|
|
153
|
+
code=EVALUATION_OUTPUT_MISSING,
|
|
154
|
+
details={"path": str(path)},
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
_refuse_non_finite_records(raw, path)
|
|
158
|
+
|
|
159
|
+
try:
|
|
160
|
+
return read_normalized_episodes(path)
|
|
161
|
+
except ValidationError as error:
|
|
162
|
+
raise ValidationError(
|
|
163
|
+
f"this variant's normalized evaluation output cannot be read: {error}",
|
|
164
|
+
code=EVALUATION_OUTPUT_CORRUPT,
|
|
165
|
+
details=dict(error.details),
|
|
166
|
+
) from error
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _refuse_non_finite_records(raw: bytes, path: Path) -> None:
|
|
170
|
+
"""Refuse a record carrying a number JSON cannot honestly hold.
|
|
171
|
+
|
|
172
|
+
``NaN`` and the infinities are spellable in the JSON Python accepts and are
|
|
173
|
+
not spellable in canonical JSON, so a record carrying one would parse here
|
|
174
|
+
and then fail, confusingly, at the moment a receipt was digested. It is
|
|
175
|
+
caught at the door instead, and a non-finite *reward* is reported as the
|
|
176
|
+
reward failure it is rather than as generic corruption.
|
|
177
|
+
"""
|
|
178
|
+
for number, line in enumerate(
|
|
179
|
+
raw.decode("utf-8", errors="replace").splitlines(), 1
|
|
180
|
+
):
|
|
181
|
+
if not line.strip():
|
|
182
|
+
continue
|
|
183
|
+
try:
|
|
184
|
+
document = json.loads(line)
|
|
185
|
+
except json.JSONDecodeError:
|
|
186
|
+
# Reported with its line number by the parser below, which knows
|
|
187
|
+
# the record shape and can say what was expected.
|
|
188
|
+
return
|
|
189
|
+
pointer = _first_non_finite(document, "")
|
|
190
|
+
if pointer is None:
|
|
191
|
+
continue
|
|
192
|
+
reward = f"/{_REWARD_KEY}/" in pointer
|
|
193
|
+
raise ValidationError(
|
|
194
|
+
(
|
|
195
|
+
f"the reward at {pointer} is not a finite number"
|
|
196
|
+
if reward
|
|
197
|
+
else f"the value at {pointer} is not a finite number"
|
|
198
|
+
),
|
|
199
|
+
code=REWARD_NON_FINITE if reward else EVALUATION_OUTPUT_CORRUPT,
|
|
200
|
+
details={"path": str(path), "line": number, "pointer": pointer},
|
|
201
|
+
)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _first_non_finite(value: object, pointer: str) -> str | None:
|
|
205
|
+
"""Return the JSON Pointer of the first non-finite number, or nothing."""
|
|
206
|
+
if isinstance(value, bool):
|
|
207
|
+
return None
|
|
208
|
+
if isinstance(value, float) and not math.isfinite(value):
|
|
209
|
+
return pointer or "/"
|
|
210
|
+
if isinstance(value, dict):
|
|
211
|
+
for key, item in value.items():
|
|
212
|
+
found = _first_non_finite(item, f"{pointer}/{key}")
|
|
213
|
+
if found is not None:
|
|
214
|
+
return found
|
|
215
|
+
return None
|
|
216
|
+
if isinstance(value, list):
|
|
217
|
+
for index, item in enumerate(value):
|
|
218
|
+
found = _first_non_finite(item, f"{pointer}/{index}")
|
|
219
|
+
if found is not None:
|
|
220
|
+
return found
|
|
221
|
+
return None
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
# ---------------------------------------------------------------------------
|
|
225
|
+
# One receipt
|
|
226
|
+
# ---------------------------------------------------------------------------
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def build_episode_receipt(
|
|
230
|
+
*,
|
|
231
|
+
run_request: RunRequest,
|
|
232
|
+
variant: VariantName,
|
|
233
|
+
experiment: ExperimentManifest,
|
|
234
|
+
episode: NormalizedEpisode,
|
|
235
|
+
raw_artifacts: VariantExecutionResult,
|
|
236
|
+
evaluation_backend: EvaluationBackendSpec,
|
|
237
|
+
primary_reward: str,
|
|
238
|
+
evidence: EvidenceRequirements,
|
|
239
|
+
) -> EpisodeReceipt:
|
|
240
|
+
"""Construct one immutable Techtree receipt from one normalized Episode.
|
|
241
|
+
|
|
242
|
+
``primary_reward`` and ``evidence`` are the two Campaign facts a receipt's
|
|
243
|
+
*statuses* depend on: which reward the comparison turns on, and whether the
|
|
244
|
+
Campaign requires runtime evidence this release cannot collect. They are
|
|
245
|
+
passed explicitly rather than by handing this function the whole Campaign,
|
|
246
|
+
because everything else a receipt cites comes from the run's request and
|
|
247
|
+
the run's manifest, and a second source for those would be a second truth.
|
|
248
|
+
"""
|
|
249
|
+
_require_lineage(
|
|
250
|
+
run_request=run_request,
|
|
251
|
+
variant=variant,
|
|
252
|
+
experiment=experiment,
|
|
253
|
+
raw_artifacts=raw_artifacts,
|
|
254
|
+
evaluation_backend=evaluation_backend,
|
|
255
|
+
)
|
|
256
|
+
trace = _subject_trace(episode, variant)
|
|
257
|
+
task_hash = validate_digest(episode.task_hash)
|
|
258
|
+
if validate_digest(trace.task_hash) != task_hash:
|
|
259
|
+
raise VerificationError(
|
|
260
|
+
"this episode's subject trace scored a different task than the "
|
|
261
|
+
"episode it belongs to",
|
|
262
|
+
code=TASK_MEMBERSHIP_MISMATCH,
|
|
263
|
+
details={"episode": episode.task_hash, "trace": trace.task_hash},
|
|
264
|
+
)
|
|
265
|
+
|
|
266
|
+
episode_digest = validate_digest(episode.raw_episode_digest)
|
|
267
|
+
return EpisodeReceipt(
|
|
268
|
+
schema_version=EPISODE_RECEIPT_SCHEMA_VERSION,
|
|
269
|
+
id=_receipt_id(
|
|
270
|
+
run_id=run_request.run_id, variant=variant, episode_digest=episode_digest
|
|
271
|
+
),
|
|
272
|
+
run_id=run_request.run_id,
|
|
273
|
+
campaign_spec_digest=run_request.campaign_spec_digest,
|
|
274
|
+
program_ref=run_request.program_ref,
|
|
275
|
+
public_context=run_request.public_context,
|
|
276
|
+
data_policy_digest=run_request.data_policy_digest,
|
|
277
|
+
outcome_contract_digest=run_request.outcome_contract_digest,
|
|
278
|
+
evaluation_backend=evaluation_backend,
|
|
279
|
+
subject_runtime=_subject_runtime(trace),
|
|
280
|
+
variant=experiment_variant_of(variant),
|
|
281
|
+
experiment_manifest_digest=raw_artifacts.experiment_manifest_digest,
|
|
282
|
+
episode_id=episode.episode_id,
|
|
283
|
+
episode_digest=episode_digest,
|
|
284
|
+
task_hash=task_hash,
|
|
285
|
+
named_traces={
|
|
286
|
+
SUBJECT_AGENT: [
|
|
287
|
+
NamedTraceReceipt(
|
|
288
|
+
role=SUBJECT_AGENT,
|
|
289
|
+
trace_id=trace.trace_id,
|
|
290
|
+
trace_digest=validate_digest(trace.raw_trace_digest),
|
|
291
|
+
task_hash=task_hash,
|
|
292
|
+
rewards=_recorded_rewards(trace),
|
|
293
|
+
metrics=_recorded_metrics(trace),
|
|
294
|
+
ok=trace.ok,
|
|
295
|
+
)
|
|
296
|
+
]
|
|
297
|
+
},
|
|
298
|
+
score_status=_score_status(episode, trace, primary_reward),
|
|
299
|
+
evidence_status=_evidence_status(trace, evidence),
|
|
300
|
+
execution_backend="verifiers",
|
|
301
|
+
artifacts=_variant_artifacts(raw_artifacts),
|
|
302
|
+
)
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def _require_lineage(
|
|
306
|
+
*,
|
|
307
|
+
run_request: RunRequest,
|
|
308
|
+
variant: VariantName,
|
|
309
|
+
experiment: ExperimentManifest,
|
|
310
|
+
raw_artifacts: VariantExecutionResult,
|
|
311
|
+
evaluation_backend: EvaluationBackendSpec,
|
|
312
|
+
) -> None:
|
|
313
|
+
"""Require every reference a receipt copies to agree with every other."""
|
|
314
|
+
expected_variant = experiment_variant_of(variant)
|
|
315
|
+
configuration = experiment.configuration
|
|
316
|
+
declared_manifest = (
|
|
317
|
+
run_request.baseline_manifest_digest
|
|
318
|
+
if expected_variant is ExperimentVariant.BASELINE
|
|
319
|
+
else run_request.candidate_manifest_digest
|
|
320
|
+
)
|
|
321
|
+
manifest_digest = digest_object(experiment)
|
|
322
|
+
|
|
323
|
+
_require(
|
|
324
|
+
raw_artifacts.variant is variant,
|
|
325
|
+
f"a {variant.value} receipt cannot be built from a "
|
|
326
|
+
f"{raw_artifacts.variant.value} execution",
|
|
327
|
+
variant=variant.value,
|
|
328
|
+
)
|
|
329
|
+
_require(
|
|
330
|
+
experiment.variant is expected_variant,
|
|
331
|
+
f"a {variant.value} receipt cannot be built from a "
|
|
332
|
+
f"{experiment.variant.value} manifest",
|
|
333
|
+
variant=variant.value,
|
|
334
|
+
)
|
|
335
|
+
_require(
|
|
336
|
+
manifest_digest == raw_artifacts.experiment_manifest_digest,
|
|
337
|
+
"this execution was produced from a different experiment manifest "
|
|
338
|
+
"than the one the receipt is being built against",
|
|
339
|
+
expected=raw_artifacts.experiment_manifest_digest,
|
|
340
|
+
computed=manifest_digest,
|
|
341
|
+
)
|
|
342
|
+
_require(
|
|
343
|
+
manifest_digest == declared_manifest,
|
|
344
|
+
f"the staged {variant.value} manifest is not the one this run's request names",
|
|
345
|
+
expected=declared_manifest,
|
|
346
|
+
computed=manifest_digest,
|
|
347
|
+
)
|
|
348
|
+
_require(
|
|
349
|
+
experiment.campaign_spec_digest == run_request.campaign_spec_digest,
|
|
350
|
+
"the manifest and the run's request name different Campaigns",
|
|
351
|
+
expected=run_request.campaign_spec_digest,
|
|
352
|
+
computed=experiment.campaign_spec_digest,
|
|
353
|
+
)
|
|
354
|
+
_require(
|
|
355
|
+
configuration.data_policy_digest == run_request.data_policy_digest,
|
|
356
|
+
"the manifest and the run's request name different DataPolicies",
|
|
357
|
+
expected=run_request.data_policy_digest,
|
|
358
|
+
computed=configuration.data_policy_digest,
|
|
359
|
+
)
|
|
360
|
+
_require(
|
|
361
|
+
configuration.outcome_contract_digest == run_request.outcome_contract_digest,
|
|
362
|
+
"the manifest and the run's request name different OutcomeContracts",
|
|
363
|
+
)
|
|
364
|
+
_require(
|
|
365
|
+
experiment.program_ref == run_request.program_ref
|
|
366
|
+
and experiment.public_context == run_request.public_context,
|
|
367
|
+
"the manifest and the run's request name a different improvement "
|
|
368
|
+
"program or public context",
|
|
369
|
+
)
|
|
370
|
+
_require(
|
|
371
|
+
evaluation_backend == run_request.evaluation_backend
|
|
372
|
+
and evaluation_backend == configuration.evaluation_backend,
|
|
373
|
+
"the evaluation backend this receipt would carry is not the one the "
|
|
374
|
+
"run's request and manifest were executed under",
|
|
375
|
+
)
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
def _require(condition: bool, message: str, **details: str) -> None:
|
|
379
|
+
"""Raise a typed lineage refusal unless the condition holds."""
|
|
380
|
+
if condition:
|
|
381
|
+
return
|
|
382
|
+
reported: dict[str, JsonValue] = dict(details)
|
|
383
|
+
raise VerificationError(message, code=EPISODE_RECEIPT_INVALID, details=reported)
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
def _subject_trace(episode: NormalizedEpisode, variant: VariantName) -> NormalizedTrace:
|
|
387
|
+
"""Return the one subject rollout this episode is allowed to carry."""
|
|
388
|
+
traces = [trace for trace in episode.traces if trace.agent_role == SUBJECT_AGENT]
|
|
389
|
+
if len(traces) == 1 and len(episode.traces) == 1:
|
|
390
|
+
return traces[0]
|
|
391
|
+
raise VerificationError(
|
|
392
|
+
f"episode {episode.episode_id} carries {len(episode.traces)} trace(s), "
|
|
393
|
+
f"{len(traces)} of them in the {SUBJECT_AGENT!r} seat; a v0.1 episode "
|
|
394
|
+
"carries exactly one subject trace and nothing else",
|
|
395
|
+
code=TRACE_ROLE_MISMATCH,
|
|
396
|
+
details={
|
|
397
|
+
"variant": variant.value,
|
|
398
|
+
"episode_id": episode.episode_id,
|
|
399
|
+
"traces": len(episode.traces),
|
|
400
|
+
"subject_traces": len(traces),
|
|
401
|
+
},
|
|
402
|
+
)
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
def _recorded_rewards(trace: NormalizedTrace) -> dict[str, float]:
|
|
406
|
+
"""Copy every reward exactly as Verifiers scored it.
|
|
407
|
+
|
|
408
|
+
The score, not the weighted value. A weight is the Campaign's opinion about
|
|
409
|
+
how much a reward should count; the score is what the evaluation measured,
|
|
410
|
+
and a receipt records the measurement. The weight and the weighted value
|
|
411
|
+
both survive in the normalized episode the receipt's artifacts point at.
|
|
412
|
+
"""
|
|
413
|
+
rewards: dict[str, float] = {}
|
|
414
|
+
for reward in trace.rewards:
|
|
415
|
+
_require_finite(reward.score, f"reward {reward.name!r}", trace)
|
|
416
|
+
rewards[reward.name] = reward.score
|
|
417
|
+
return rewards
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def _recorded_metrics(trace: NormalizedTrace) -> dict[str, float | None]:
|
|
421
|
+
"""Copy every metric exactly as recorded."""
|
|
422
|
+
for name, value in trace.metrics.items():
|
|
423
|
+
if value is not None:
|
|
424
|
+
_require_finite(value, f"metric {name!r}", trace, reward=False)
|
|
425
|
+
return dict(trace.metrics)
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
def _require_finite(
|
|
429
|
+
value: float, label: str, trace: NormalizedTrace, *, reward: bool = True
|
|
430
|
+
) -> None:
|
|
431
|
+
"""Refuse a number a receipt could not be digested with."""
|
|
432
|
+
if math.isfinite(value):
|
|
433
|
+
return
|
|
434
|
+
raise VerificationError(
|
|
435
|
+
f"the {label} recorded on trace {trace.trace_id} is not finite, so it "
|
|
436
|
+
"is not a measurement",
|
|
437
|
+
code=REWARD_NON_FINITE if reward else EVALUATION_OUTPUT_CORRUPT,
|
|
438
|
+
details={"trace_id": trace.trace_id, "value": repr(value)},
|
|
439
|
+
)
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
def _subject_runtime(trace: NormalizedTrace) -> SubjectRuntimeReceipt:
|
|
443
|
+
"""Describe the box the subject actually executed in.
|
|
444
|
+
|
|
445
|
+
A Docker receipt must cite the image it ran, and it cites the content the
|
|
446
|
+
pinned reference names. Which platform that content was served on is a fact
|
|
447
|
+
about the machine rather than about the episode, so it lives in the
|
|
448
|
+
comparison's observed configuration and is left unset here.
|
|
449
|
+
"""
|
|
450
|
+
return SubjectRuntimeReceipt(
|
|
451
|
+
kind="docker",
|
|
452
|
+
resolved_image_digest=validate_digest(trace.runtime.image_index_digest),
|
|
453
|
+
platform=None,
|
|
454
|
+
)
|
|
455
|
+
|
|
456
|
+
|
|
457
|
+
def _score_status(
|
|
458
|
+
episode: NormalizedEpisode, trace: NormalizedTrace, primary_reward: str
|
|
459
|
+
) -> ScoreStatus:
|
|
460
|
+
"""Say how much weight the recorded reward carries. Spec section 7.6."""
|
|
461
|
+
if not episode.ok or not trace.ok or episode.errors or trace.errors:
|
|
462
|
+
return ScoreStatus.ERRORED
|
|
463
|
+
if trace.reward(primary_reward) is None:
|
|
464
|
+
return ScoreStatus.MISSING
|
|
465
|
+
return ScoreStatus.VALID
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
def _evidence_status(
|
|
469
|
+
trace: NormalizedTrace, evidence: EvidenceRequirements
|
|
470
|
+
) -> EvidenceStatus:
|
|
471
|
+
"""Say how complete the supporting evidence is. Spec section 7.6.
|
|
472
|
+
|
|
473
|
+
Every artifact reference a receipt carries is a required field of the
|
|
474
|
+
execution result, hashed from the bytes on the way in, so their presence is
|
|
475
|
+
a property of the object rather than something to re-check. What is left to
|
|
476
|
+
decide is whether the subject configuration was recorded at all and whether
|
|
477
|
+
the Campaign asked for runtime evidence this release does not collect —
|
|
478
|
+
absence of a Relay is not incompleteness when the Campaign says runtime
|
|
479
|
+
evidence is not required.
|
|
480
|
+
"""
|
|
481
|
+
configured = bool(trace.model_id and trace.harness_id and trace.tools)
|
|
482
|
+
if configured and evidence.runtime_evidence != "required":
|
|
483
|
+
return EvidenceStatus.COMPLETE
|
|
484
|
+
return EvidenceStatus.PARTIAL
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
def _variant_artifacts(result: VariantExecutionResult) -> list[ArtifactRef]:
|
|
488
|
+
"""Reference the variant's evidence once, in one fixed order.
|
|
489
|
+
|
|
490
|
+
Spec section 7.6 permits a receipt to point at the variant's artifacts
|
|
491
|
+
rather than duplicating them per episode, and every one of these is a whole
|
|
492
|
+
file covering the variant: the configuration the engine resolved, the raw
|
|
493
|
+
upstream record, the engine's log, and the normalized projection this
|
|
494
|
+
receipt was built from.
|
|
495
|
+
"""
|
|
496
|
+
return [
|
|
497
|
+
result.resolved_verifiers_config,
|
|
498
|
+
result.raw_traces,
|
|
499
|
+
result.eval_log,
|
|
500
|
+
result.normalized_episodes,
|
|
501
|
+
]
|
|
502
|
+
|
|
503
|
+
|
|
504
|
+
def _receipt_id(*, run_id: str, variant: VariantName, episode_digest: Digest) -> str:
|
|
505
|
+
"""Return a deterministic identifier in the shape :mod:`techtree.ids` issues.
|
|
506
|
+
|
|
507
|
+
Derived rather than random so that rebuilding a receipt from the same
|
|
508
|
+
evidence produces the same identifier, which is what lets a rebuild be
|
|
509
|
+
compared to what was stored.
|
|
510
|
+
"""
|
|
511
|
+
digest = sha256_digest_bytes(
|
|
512
|
+
canonical_json_bytes(
|
|
513
|
+
{
|
|
514
|
+
"run_id": run_id,
|
|
515
|
+
"variant": variant.value,
|
|
516
|
+
"episode_digest": episode_digest,
|
|
517
|
+
}
|
|
518
|
+
)
|
|
519
|
+
)
|
|
520
|
+
_, _, hexadecimal = digest.partition(":")
|
|
521
|
+
return f"receipt_{hexadecimal[:_DERIVED_ID_LENGTH]}"
|
|
522
|
+
|
|
523
|
+
|
|
524
|
+
# ---------------------------------------------------------------------------
|
|
525
|
+
# One variant's receipts
|
|
526
|
+
# ---------------------------------------------------------------------------
|
|
527
|
+
|
|
528
|
+
|
|
529
|
+
def build_variant_receipts(
|
|
530
|
+
*,
|
|
531
|
+
run_request: RunRequest,
|
|
532
|
+
variant: VariantName,
|
|
533
|
+
experiment: ExperimentManifest,
|
|
534
|
+
result: VariantExecutionResult,
|
|
535
|
+
evaluation_backend: EvaluationBackendSpec,
|
|
536
|
+
ordered_task_hashes: Sequence[Digest],
|
|
537
|
+
primary_reward: str,
|
|
538
|
+
evidence: EvidenceRequirements,
|
|
539
|
+
) -> list[EpisodeReceipt]:
|
|
540
|
+
"""Build exactly one receipt per committed task, in committed order.
|
|
541
|
+
|
|
542
|
+
The join is on task hash and never on position in a file. Completion order
|
|
543
|
+
varies between two concurrent variants and has nothing to do with the
|
|
544
|
+
Campaign's commitment, so a receipt list built by walking the file would be
|
|
545
|
+
a list nobody could pair.
|
|
546
|
+
|
|
547
|
+
A variant with an unscored task is refused. One rollout that completed
|
|
548
|
+
cleanly and produced no primary reward means scoring did not run, and a
|
|
549
|
+
comparison computed over the tasks that happened to score would be a
|
|
550
|
+
comparison over a taskset the Campaign never committed to.
|
|
551
|
+
|
|
552
|
+
The lineage is settled before the episodes are looked at, so that an
|
|
553
|
+
execution filed under the wrong variant is reported as the provenance
|
|
554
|
+
failure it is rather than as whichever counting rule it happens to trip
|
|
555
|
+
first.
|
|
556
|
+
"""
|
|
557
|
+
_require_lineage(
|
|
558
|
+
run_request=run_request,
|
|
559
|
+
variant=variant,
|
|
560
|
+
experiment=experiment,
|
|
561
|
+
raw_artifacts=result,
|
|
562
|
+
evaluation_backend=evaluation_backend,
|
|
563
|
+
)
|
|
564
|
+
committed = _committed_membership(ordered_task_hashes)
|
|
565
|
+
episodes = _episodes_by_task(result, committed, variant)
|
|
566
|
+
|
|
567
|
+
receipts = [
|
|
568
|
+
build_episode_receipt(
|
|
569
|
+
run_request=run_request,
|
|
570
|
+
variant=variant,
|
|
571
|
+
experiment=experiment,
|
|
572
|
+
episode=episodes[task_hash],
|
|
573
|
+
raw_artifacts=result,
|
|
574
|
+
evaluation_backend=evaluation_backend,
|
|
575
|
+
primary_reward=primary_reward,
|
|
576
|
+
evidence=evidence,
|
|
577
|
+
)
|
|
578
|
+
for task_hash in committed
|
|
579
|
+
]
|
|
580
|
+
_require_every_task_scored(receipts, primary_reward, variant)
|
|
581
|
+
return receipts
|
|
582
|
+
|
|
583
|
+
|
|
584
|
+
def _committed_membership(ordered_task_hashes: Sequence[Digest]) -> list[Digest]:
|
|
585
|
+
"""Revalidate the membership a variant's receipts are ordered by."""
|
|
586
|
+
committed = [validate_digest(value) for value in ordered_task_hashes]
|
|
587
|
+
if not committed:
|
|
588
|
+
raise VerificationError(
|
|
589
|
+
"a Campaign commits to at least one task, and this one commits to "
|
|
590
|
+
"none, so there is nothing to build receipts for",
|
|
591
|
+
code=TASK_MEMBERSHIP_MISMATCH,
|
|
592
|
+
details={"task_count": 0},
|
|
593
|
+
)
|
|
594
|
+
if len(set(committed)) != len(committed):
|
|
595
|
+
raise VerificationError(
|
|
596
|
+
"the committed membership names the same task twice, so a receipt "
|
|
597
|
+
"could be attributed to either position",
|
|
598
|
+
code=TASK_MEMBERSHIP_MISMATCH,
|
|
599
|
+
details={"task_count": len(committed)},
|
|
600
|
+
)
|
|
601
|
+
return committed
|
|
602
|
+
|
|
603
|
+
|
|
604
|
+
def _episodes_by_task(
|
|
605
|
+
result: VariantExecutionResult,
|
|
606
|
+
committed: Sequence[Digest],
|
|
607
|
+
variant: VariantName,
|
|
608
|
+
) -> dict[Digest, NormalizedEpisode]:
|
|
609
|
+
"""Index one variant's episodes by task, or say why they cannot be joined."""
|
|
610
|
+
if len(result.episodes) != len(committed):
|
|
611
|
+
raise VerificationError(
|
|
612
|
+
f"the {variant.value} variant recorded {len(result.episodes)} "
|
|
613
|
+
f"episodes for {len(committed)} committed tasks",
|
|
614
|
+
code=EPISODE_COUNT_MISMATCH,
|
|
615
|
+
details={
|
|
616
|
+
"variant": variant.value,
|
|
617
|
+
"recorded": len(result.episodes),
|
|
618
|
+
"committed": len(committed),
|
|
619
|
+
},
|
|
620
|
+
)
|
|
621
|
+
|
|
622
|
+
by_task: dict[Digest, NormalizedEpisode] = {}
|
|
623
|
+
for episode in result.episodes:
|
|
624
|
+
task_hash = validate_digest(episode.task_hash)
|
|
625
|
+
if task_hash in by_task:
|
|
626
|
+
raise VerificationError(
|
|
627
|
+
f"the {variant.value} variant scored task {task_hash} twice, so "
|
|
628
|
+
"one of the two results would have to be discarded",
|
|
629
|
+
code=TASK_MEMBERSHIP_MISMATCH,
|
|
630
|
+
details={"variant": variant.value, "task_hash": task_hash},
|
|
631
|
+
)
|
|
632
|
+
by_task[task_hash] = episode
|
|
633
|
+
|
|
634
|
+
missing: list[JsonValue] = [value for value in committed if value not in by_task]
|
|
635
|
+
unexpected: list[JsonValue] = [
|
|
636
|
+
value for value in sorted(set(by_task) - set(committed))
|
|
637
|
+
]
|
|
638
|
+
if missing or unexpected:
|
|
639
|
+
raise VerificationError(
|
|
640
|
+
f"the {variant.value} variant scored a different set of tasks than "
|
|
641
|
+
"the Campaign commits to",
|
|
642
|
+
code=TASK_MEMBERSHIP_MISMATCH,
|
|
643
|
+
details={
|
|
644
|
+
"variant": variant.value,
|
|
645
|
+
"missing": missing,
|
|
646
|
+
"unexpected": unexpected,
|
|
647
|
+
},
|
|
648
|
+
)
|
|
649
|
+
return by_task
|
|
650
|
+
|
|
651
|
+
|
|
652
|
+
def _require_every_task_scored(
|
|
653
|
+
receipts: Sequence[EpisodeReceipt], primary_reward: str, variant: VariantName
|
|
654
|
+
) -> None:
|
|
655
|
+
"""Refuse a variant whose clean rollouts produced no primary reward."""
|
|
656
|
+
unscored: list[JsonValue] = [
|
|
657
|
+
receipt.task_hash
|
|
658
|
+
for receipt in receipts
|
|
659
|
+
if receipt.score_status is ScoreStatus.MISSING
|
|
660
|
+
]
|
|
661
|
+
if not unscored:
|
|
662
|
+
return
|
|
663
|
+
raise VerificationError(
|
|
664
|
+
f"{len(unscored)} {variant.value} rollout(s) completed without scoring "
|
|
665
|
+
f"{primary_reward!r}, which is the reward this comparison is decided on",
|
|
666
|
+
code=REWARD_MISSING,
|
|
667
|
+
details={
|
|
668
|
+
"variant": variant.value,
|
|
669
|
+
"reward": primary_reward,
|
|
670
|
+
"task_hashes": unscored,
|
|
671
|
+
},
|
|
672
|
+
)
|