techtree 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. techtree/__init__.py +35 -0
  2. techtree/__main__.py +14 -0
  3. techtree/canonical.py +239 -0
  4. techtree/catalog/__init__.py +25 -0
  5. techtree/catalog/repository.py +400 -0
  6. techtree/catalog/service.py +419 -0
  7. techtree/cli/__init__.py +1 -0
  8. techtree/cli/app.py +416 -0
  9. techtree/cli/commands/__init__.py +1 -0
  10. techtree/cli/commands/climb.py +1223 -0
  11. techtree/cli/commands/doctor.py +147 -0
  12. techtree/cli/commands/engine.py +207 -0
  13. techtree/cli/commands/proof.py +556 -0
  14. techtree/cli/commands/publish.py +447 -0
  15. techtree/cli/commands/release.py +303 -0
  16. techtree/cli/commands/run.py +1067 -0
  17. techtree/cli/commands/setup.py +181 -0
  18. techtree/cli/commands/skill.py +221 -0
  19. techtree/cli/commands/uplift.py +698 -0
  20. techtree/cli/commands/withdraw.py +212 -0
  21. techtree/cli/confirm.py +47 -0
  22. techtree/cli/context.py +96 -0
  23. techtree/cli/invoke.py +220 -0
  24. techtree/cli/output.py +280 -0
  25. techtree/constants.py +138 -0
  26. techtree/crypto.py +128 -0
  27. techtree/doctor/__init__.py +1 -0
  28. techtree/doctor/checks.py +675 -0
  29. techtree/doctor/execution_checks.py +435 -0
  30. techtree/doctor/service.py +326 -0
  31. techtree/drafts/__init__.py +32 -0
  32. techtree/drafts/source.py +146 -0
  33. techtree/drafts/store.py +992 -0
  34. techtree/engines/__init__.py +1 -0
  35. techtree/engines/bundle.py +251 -0
  36. techtree/engines/installer.py +679 -0
  37. techtree/engines/registry.py +235 -0
  38. techtree/engines/runner.py +170 -0
  39. techtree/errors.py +262 -0
  40. techtree/fs.py +234 -0
  41. techtree/harness.py +108 -0
  42. techtree/identity/__init__.py +41 -0
  43. techtree/identity/models.py +113 -0
  44. techtree/identity/service.py +199 -0
  45. techtree/identity/store.py +263 -0
  46. techtree/ids.py +85 -0
  47. techtree/manifests/__init__.py +39 -0
  48. techtree/manifests/builder.py +433 -0
  49. techtree/manifests/compare.py +376 -0
  50. techtree/models/__init__.py +282 -0
  51. techtree/models/base.py +201 -0
  52. techtree/models/campaign.py +484 -0
  53. techtree/models/catalog.py +227 -0
  54. techtree/models/cli.py +151 -0
  55. techtree/models/climb.py +254 -0
  56. techtree/models/data_policy.py +130 -0
  57. techtree/models/engine.py +156 -0
  58. techtree/models/episode_receipt.py +130 -0
  59. techtree/models/evaluation_backend.py +113 -0
  60. techtree/models/experiment.py +154 -0
  61. techtree/models/run.py +214 -0
  62. techtree/models/skill.py +156 -0
  63. techtree/models/uplift_report.py +158 -0
  64. techtree/models/validation.py +299 -0
  65. techtree/paths.py +116 -0
  66. techtree/presentation/__init__.py +31 -0
  67. techtree/presentation/build.py +1242 -0
  68. techtree/presentation/compact.py +246 -0
  69. techtree/presentation/evidence.py +169 -0
  70. techtree/presentation/models.py +358 -0
  71. techtree/presentation/rich.py +312 -0
  72. techtree/presentation/sanitize.py +156 -0
  73. techtree/publication/__init__.py +44 -0
  74. techtree/publication/address.py +180 -0
  75. techtree/publication/coordinates.py +26 -0
  76. techtree/publication/journal.py +212 -0
  77. techtree/publication/keccak.py +183 -0
  78. techtree/publication/models.py +209 -0
  79. techtree/publication/offer.py +35 -0
  80. techtree/publication/service.py +618 -0
  81. techtree/publication/transport.py +296 -0
  82. techtree/publication/verify.py +242 -0
  83. techtree/publication/withdraw.py +156 -0
  84. techtree/py.typed +0 -0
  85. techtree/receipts/__init__.py +52 -0
  86. techtree/receipts/bundle.py +578 -0
  87. techtree/receipts/compare.py +1065 -0
  88. techtree/receipts/episode.py +672 -0
  89. techtree/receipts/execution.py +630 -0
  90. techtree/receipts/observed.py +474 -0
  91. techtree/receipts/set.py +336 -0
  92. techtree/receipts/uplift.py +655 -0
  93. techtree/receipts/verify.py +1055 -0
  94. techtree/release/__init__.py +9 -0
  95. techtree/release/bootstrap.py +509 -0
  96. techtree/release/checks.py +376 -0
  97. techtree/release/document.py +125 -0
  98. techtree/release/generate.py +221 -0
  99. techtree/release/models.py +293 -0
  100. techtree/release/provenance.py +109 -0
  101. techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
  102. techtree/resources/catalog/catalog.json +32 -0
  103. techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
  104. techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
  105. techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
  106. techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
  107. techtree/resources/engines/default/engine.json +20 -0
  108. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
  109. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
  110. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
  111. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
  112. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
  113. techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
  114. techtree/resources/engines/default/pyproject.toml +23 -0
  115. techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
  116. techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
  117. techtree/resources/engines/default/tools/normalize_validation.py +222 -0
  118. techtree/resources/engines/default/uv.lock +1758 -0
  119. techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
  120. techtree/resources/release/build-provenance.json +4 -0
  121. techtree/resources/release/release-core.json +24 -0
  122. techtree/runs/__init__.py +31 -0
  123. techtree/runs/artifacts.py +750 -0
  124. techtree/runs/child_registry.py +228 -0
  125. techtree/runs/events.py +478 -0
  126. techtree/runs/executor.py +140 -0
  127. techtree/runs/fake.py +741 -0
  128. techtree/runs/launcher.py +253 -0
  129. techtree/runs/machine.py +489 -0
  130. techtree/runs/real.py +789 -0
  131. techtree/runs/service.py +616 -0
  132. techtree/runs/store.py +555 -0
  133. techtree/runs/validation.py +259 -0
  134. techtree/runs/variants.py +684 -0
  135. techtree/settings.py +143 -0
  136. techtree/skills/__init__.py +14 -0
  137. techtree/skills/archive.py +282 -0
  138. techtree/skills/policy.py +62 -0
  139. techtree/skills/scanner.py +394 -0
  140. techtree/skills/service.py +752 -0
  141. techtree/skills/starter.py +434 -0
  142. techtree/tasksets/__init__.py +1 -0
  143. techtree/tasksets/membership.py +269 -0
  144. techtree/tasksets/provider.py +207 -0
  145. techtree/tasksets/resolver.py +311 -0
  146. techtree/tasksets/service.py +484 -0
  147. techtree/tasksets/verifiers_cli.py +538 -0
  148. techtree/uplift/__init__.py +20 -0
  149. techtree/uplift/context.py +544 -0
  150. techtree/uplift/derive.py +203 -0
  151. techtree/uplift/public_tasks.py +151 -0
  152. techtree/uplift/service.py +719 -0
  153. techtree/uplift/source.py +160 -0
  154. techtree/verifiers/__init__.py +31 -0
  155. techtree/verifiers/budget.py +219 -0
  156. techtree/verifiers/child.py +633 -0
  157. techtree/verifiers/compiler.py +432 -0
  158. techtree/verifiers/config.py +365 -0
  159. techtree/verifiers/credentials.py +321 -0
  160. techtree/verifiers/image.py +126 -0
  161. techtree/verifiers/models.py +527 -0
  162. techtree/verifiers/outputs.py +368 -0
  163. techtree/verifiers/progress.py +192 -0
  164. techtree/verifiers/supervisor.py +341 -0
  165. techtree/verifiers/verify.py +782 -0
  166. techtree/version.py +39 -0
  167. techtree/worker/__init__.py +18 -0
  168. techtree/worker/execute.py +487 -0
  169. techtree/worker/main.py +57 -0
  170. techtree-0.1.0.dist-info/METADATA +344 -0
  171. techtree-0.1.0.dist-info/RECORD +174 -0
  172. techtree-0.1.0.dist-info/WHEEL +4 -0
  173. techtree-0.1.0.dist-info/entry_points.txt +3 -0
  174. techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,672 @@
1
+ """One receipt per committed task, from recorded evidence. Spec section 7.6.
2
+
3
+ The input is the engine's own normalized projection of one variant's episodes
4
+ and the run's immutable lineage. The output is one
5
+ :class:`~techtree.models.episode_receipt.EpisodeReceipt` per task the Campaign
6
+ committed to, in committed order, carrying the reward Verifiers recorded and
7
+ nothing derived from it.
8
+
9
+ Three boundaries meet in this module and each is drawn deliberately.
10
+
11
+ *The evidence boundary.* :func:`read_variant_episodes` is the only door the
12
+ normalized file comes through, and it maps every way that file can be unusable
13
+ onto the spec section 15 vocabulary: absent evidence is
14
+ ``evaluation_output_missing``, unreadable evidence is
15
+ ``evaluation_output_corrupt``, and a reward that is not a number is
16
+ ``reward_non_finite`` rather than a generic parse failure — a run whose scoring
17
+ produced ``NaN`` failed in a specific way and a reader deserves to be told
18
+ which.
19
+
20
+ *The task-hash boundary.* Verifiers spells a task hash as bare hexadecimal and
21
+ Techtree spells it as a prefixed digest. The engine's normalizer converts it
22
+ once, on the side that read the wire record, and every hash that enters this
23
+ module is revalidated as a Techtree digest before it is compared to anything.
24
+ A type annotation is a claim about the caller; a commitment other parties are
25
+ held to is checked.
26
+
27
+ *The lineage boundary.* A receipt names the Campaign, the improvement program,
28
+ the public context, the DataPolicy, the OutcomeContract, the evaluation backend
29
+ and the experiment manifest it was produced under. Every one of those is copied
30
+ from the run's own immutable request and the run's own staged manifest, and
31
+ every one is required to agree with the others before a receipt is built.
32
+ Nothing is defaulted and nothing is re-resolved.
33
+
34
+ What is *not* a refusal matters as much. An episode that failed, a trace that
35
+ errored, a rollout the provider gave up on: those produce receipts with
36
+ ``score_status`` saying so. The refusals are reserved for evidence that cannot
37
+ be joined onto the Campaign's commitment at all, because those are the cases
38
+ where continuing would produce a tidy document making a false claim.
39
+ """
40
+
41
+ from __future__ import annotations
42
+
43
+ import json
44
+ import math
45
+ from collections.abc import Sequence
46
+ from pathlib import Path
47
+ from typing import Final
48
+
49
+ from techtree.canonical import (
50
+ canonical_json_bytes,
51
+ digest_object,
52
+ sha256_digest_bytes,
53
+ validate_digest,
54
+ )
55
+ from techtree.constants import EPISODE_RECEIPT_SCHEMA_VERSION
56
+ from techtree.errors import ValidationError, VerificationError
57
+ from techtree.models.base import ArtifactRef, Digest, JsonValue
58
+ from techtree.models.campaign import SUBJECT_AGENT, EvidenceRequirements
59
+ from techtree.models.episode_receipt import (
60
+ EpisodeReceipt,
61
+ EvidenceStatus,
62
+ NamedTraceReceipt,
63
+ ScoreStatus,
64
+ SubjectRuntimeReceipt,
65
+ )
66
+ from techtree.models.evaluation_backend import EvaluationBackendSpec
67
+ from techtree.models.experiment import ExperimentManifest, ExperimentVariant
68
+ from techtree.models.run import RunRequest
69
+ from techtree.verifiers.models import (
70
+ NormalizedEpisode,
71
+ NormalizedTrace,
72
+ VariantExecutionResult,
73
+ VariantName,
74
+ )
75
+ from techtree.verifiers.outputs import read_normalized_episodes
76
+
77
+ __all__ = [
78
+ "EPISODE_COUNT_MISMATCH",
79
+ "EPISODE_RECEIPT_INVALID",
80
+ "EVALUATION_OUTPUT_CORRUPT",
81
+ "EVALUATION_OUTPUT_MISSING",
82
+ "REWARD_MISSING",
83
+ "REWARD_NON_FINITE",
84
+ "TASK_MEMBERSHIP_MISMATCH",
85
+ "TRACE_ROLE_MISMATCH",
86
+ "build_episode_receipt",
87
+ "build_variant_receipts",
88
+ "experiment_variant_of",
89
+ "read_variant_episodes",
90
+ ]
91
+
92
+ #: Stable error codes. Spec section 15 fixes the vocabulary; this module is
93
+ #: where the evidence-side half of it is raised.
94
+ EVALUATION_OUTPUT_MISSING: Final = "evaluation_output_missing"
95
+ EVALUATION_OUTPUT_CORRUPT: Final = "evaluation_output_corrupt"
96
+ EPISODE_COUNT_MISMATCH: Final = "episode_count_mismatch"
97
+ TASK_MEMBERSHIP_MISMATCH: Final = "task_membership_mismatch"
98
+ TRACE_ROLE_MISMATCH: Final = "trace_role_mismatch"
99
+ REWARD_MISSING: Final = "reward_missing"
100
+ REWARD_NON_FINITE: Final = "reward_non_finite"
101
+
102
+ #: A receipt whose lineage does not hold together. Spec section 15 lists the
103
+ #: evidence failures; this is the one that says the *provenance* disagrees,
104
+ #: which is a different question and deserves its own code.
105
+ EPISODE_RECEIPT_INVALID: Final = "episode_receipt_invalid"
106
+
107
+ #: How many hexadecimal characters a derived identifier carries, matching the
108
+ #: 32 that :mod:`techtree.ids` issues.
109
+ _DERIVED_ID_LENGTH: Final = 32
110
+
111
+ #: The JSON key that makes a non-finite number under it a reward failure
112
+ #: rather than a generic corruption.
113
+ _REWARD_KEY: Final = "rewards"
114
+
115
+
116
+ def experiment_variant_of(variant: VariantName) -> ExperimentVariant:
117
+ """Return the protocol variant one execution-local variant names.
118
+
119
+ Two enumerations spell the same two words: one belongs to the execution
120
+ layer, which schedules children, and one belongs to the protocol, which
121
+ describes manifests and receipts. Converting between them explicitly, in
122
+ one place, keeps the layers separate without letting either drift.
123
+ """
124
+ return ExperimentVariant(variant.value)
125
+
126
+
127
+ # ---------------------------------------------------------------------------
128
+ # The evidence boundary
129
+ # ---------------------------------------------------------------------------
130
+
131
+
132
+ def read_variant_episodes(path: Path) -> list[NormalizedEpisode]:
133
+ """Read one variant's normalized episodes, or say precisely what is wrong.
134
+
135
+ The raw ``traces.jsonl`` is deliberately not an accepted input here. It is
136
+ read once, by the pinned Verifiers build that wrote it, inside the engine
137
+ bundle; this is the only projection Techtree interprets.
138
+ """
139
+ try:
140
+ raw = path.read_bytes()
141
+ except OSError as error:
142
+ raise ValidationError(
143
+ "this variant's normalized evaluation output was not found, so "
144
+ "there is no evidence to build receipts from",
145
+ code=EVALUATION_OUTPUT_MISSING,
146
+ details={"path": str(path)},
147
+ ) from error
148
+
149
+ if not raw:
150
+ raise ValidationError(
151
+ "this variant's normalized evaluation output is empty, so no "
152
+ "episode was ever recorded",
153
+ code=EVALUATION_OUTPUT_MISSING,
154
+ details={"path": str(path)},
155
+ )
156
+
157
+ _refuse_non_finite_records(raw, path)
158
+
159
+ try:
160
+ return read_normalized_episodes(path)
161
+ except ValidationError as error:
162
+ raise ValidationError(
163
+ f"this variant's normalized evaluation output cannot be read: {error}",
164
+ code=EVALUATION_OUTPUT_CORRUPT,
165
+ details=dict(error.details),
166
+ ) from error
167
+
168
+
169
+ def _refuse_non_finite_records(raw: bytes, path: Path) -> None:
170
+ """Refuse a record carrying a number JSON cannot honestly hold.
171
+
172
+ ``NaN`` and the infinities are spellable in the JSON Python accepts and are
173
+ not spellable in canonical JSON, so a record carrying one would parse here
174
+ and then fail, confusingly, at the moment a receipt was digested. It is
175
+ caught at the door instead, and a non-finite *reward* is reported as the
176
+ reward failure it is rather than as generic corruption.
177
+ """
178
+ for number, line in enumerate(
179
+ raw.decode("utf-8", errors="replace").splitlines(), 1
180
+ ):
181
+ if not line.strip():
182
+ continue
183
+ try:
184
+ document = json.loads(line)
185
+ except json.JSONDecodeError:
186
+ # Reported with its line number by the parser below, which knows
187
+ # the record shape and can say what was expected.
188
+ return
189
+ pointer = _first_non_finite(document, "")
190
+ if pointer is None:
191
+ continue
192
+ reward = f"/{_REWARD_KEY}/" in pointer
193
+ raise ValidationError(
194
+ (
195
+ f"the reward at {pointer} is not a finite number"
196
+ if reward
197
+ else f"the value at {pointer} is not a finite number"
198
+ ),
199
+ code=REWARD_NON_FINITE if reward else EVALUATION_OUTPUT_CORRUPT,
200
+ details={"path": str(path), "line": number, "pointer": pointer},
201
+ )
202
+
203
+
204
+ def _first_non_finite(value: object, pointer: str) -> str | None:
205
+ """Return the JSON Pointer of the first non-finite number, or nothing."""
206
+ if isinstance(value, bool):
207
+ return None
208
+ if isinstance(value, float) and not math.isfinite(value):
209
+ return pointer or "/"
210
+ if isinstance(value, dict):
211
+ for key, item in value.items():
212
+ found = _first_non_finite(item, f"{pointer}/{key}")
213
+ if found is not None:
214
+ return found
215
+ return None
216
+ if isinstance(value, list):
217
+ for index, item in enumerate(value):
218
+ found = _first_non_finite(item, f"{pointer}/{index}")
219
+ if found is not None:
220
+ return found
221
+ return None
222
+
223
+
224
+ # ---------------------------------------------------------------------------
225
+ # One receipt
226
+ # ---------------------------------------------------------------------------
227
+
228
+
229
+ def build_episode_receipt(
230
+ *,
231
+ run_request: RunRequest,
232
+ variant: VariantName,
233
+ experiment: ExperimentManifest,
234
+ episode: NormalizedEpisode,
235
+ raw_artifacts: VariantExecutionResult,
236
+ evaluation_backend: EvaluationBackendSpec,
237
+ primary_reward: str,
238
+ evidence: EvidenceRequirements,
239
+ ) -> EpisodeReceipt:
240
+ """Construct one immutable Techtree receipt from one normalized Episode.
241
+
242
+ ``primary_reward`` and ``evidence`` are the two Campaign facts a receipt's
243
+ *statuses* depend on: which reward the comparison turns on, and whether the
244
+ Campaign requires runtime evidence this release cannot collect. They are
245
+ passed explicitly rather than by handing this function the whole Campaign,
246
+ because everything else a receipt cites comes from the run's request and
247
+ the run's manifest, and a second source for those would be a second truth.
248
+ """
249
+ _require_lineage(
250
+ run_request=run_request,
251
+ variant=variant,
252
+ experiment=experiment,
253
+ raw_artifacts=raw_artifacts,
254
+ evaluation_backend=evaluation_backend,
255
+ )
256
+ trace = _subject_trace(episode, variant)
257
+ task_hash = validate_digest(episode.task_hash)
258
+ if validate_digest(trace.task_hash) != task_hash:
259
+ raise VerificationError(
260
+ "this episode's subject trace scored a different task than the "
261
+ "episode it belongs to",
262
+ code=TASK_MEMBERSHIP_MISMATCH,
263
+ details={"episode": episode.task_hash, "trace": trace.task_hash},
264
+ )
265
+
266
+ episode_digest = validate_digest(episode.raw_episode_digest)
267
+ return EpisodeReceipt(
268
+ schema_version=EPISODE_RECEIPT_SCHEMA_VERSION,
269
+ id=_receipt_id(
270
+ run_id=run_request.run_id, variant=variant, episode_digest=episode_digest
271
+ ),
272
+ run_id=run_request.run_id,
273
+ campaign_spec_digest=run_request.campaign_spec_digest,
274
+ program_ref=run_request.program_ref,
275
+ public_context=run_request.public_context,
276
+ data_policy_digest=run_request.data_policy_digest,
277
+ outcome_contract_digest=run_request.outcome_contract_digest,
278
+ evaluation_backend=evaluation_backend,
279
+ subject_runtime=_subject_runtime(trace),
280
+ variant=experiment_variant_of(variant),
281
+ experiment_manifest_digest=raw_artifacts.experiment_manifest_digest,
282
+ episode_id=episode.episode_id,
283
+ episode_digest=episode_digest,
284
+ task_hash=task_hash,
285
+ named_traces={
286
+ SUBJECT_AGENT: [
287
+ NamedTraceReceipt(
288
+ role=SUBJECT_AGENT,
289
+ trace_id=trace.trace_id,
290
+ trace_digest=validate_digest(trace.raw_trace_digest),
291
+ task_hash=task_hash,
292
+ rewards=_recorded_rewards(trace),
293
+ metrics=_recorded_metrics(trace),
294
+ ok=trace.ok,
295
+ )
296
+ ]
297
+ },
298
+ score_status=_score_status(episode, trace, primary_reward),
299
+ evidence_status=_evidence_status(trace, evidence),
300
+ execution_backend="verifiers",
301
+ artifacts=_variant_artifacts(raw_artifacts),
302
+ )
303
+
304
+
305
+ def _require_lineage(
306
+ *,
307
+ run_request: RunRequest,
308
+ variant: VariantName,
309
+ experiment: ExperimentManifest,
310
+ raw_artifacts: VariantExecutionResult,
311
+ evaluation_backend: EvaluationBackendSpec,
312
+ ) -> None:
313
+ """Require every reference a receipt copies to agree with every other."""
314
+ expected_variant = experiment_variant_of(variant)
315
+ configuration = experiment.configuration
316
+ declared_manifest = (
317
+ run_request.baseline_manifest_digest
318
+ if expected_variant is ExperimentVariant.BASELINE
319
+ else run_request.candidate_manifest_digest
320
+ )
321
+ manifest_digest = digest_object(experiment)
322
+
323
+ _require(
324
+ raw_artifacts.variant is variant,
325
+ f"a {variant.value} receipt cannot be built from a "
326
+ f"{raw_artifacts.variant.value} execution",
327
+ variant=variant.value,
328
+ )
329
+ _require(
330
+ experiment.variant is expected_variant,
331
+ f"a {variant.value} receipt cannot be built from a "
332
+ f"{experiment.variant.value} manifest",
333
+ variant=variant.value,
334
+ )
335
+ _require(
336
+ manifest_digest == raw_artifacts.experiment_manifest_digest,
337
+ "this execution was produced from a different experiment manifest "
338
+ "than the one the receipt is being built against",
339
+ expected=raw_artifacts.experiment_manifest_digest,
340
+ computed=manifest_digest,
341
+ )
342
+ _require(
343
+ manifest_digest == declared_manifest,
344
+ f"the staged {variant.value} manifest is not the one this run's request names",
345
+ expected=declared_manifest,
346
+ computed=manifest_digest,
347
+ )
348
+ _require(
349
+ experiment.campaign_spec_digest == run_request.campaign_spec_digest,
350
+ "the manifest and the run's request name different Campaigns",
351
+ expected=run_request.campaign_spec_digest,
352
+ computed=experiment.campaign_spec_digest,
353
+ )
354
+ _require(
355
+ configuration.data_policy_digest == run_request.data_policy_digest,
356
+ "the manifest and the run's request name different DataPolicies",
357
+ expected=run_request.data_policy_digest,
358
+ computed=configuration.data_policy_digest,
359
+ )
360
+ _require(
361
+ configuration.outcome_contract_digest == run_request.outcome_contract_digest,
362
+ "the manifest and the run's request name different OutcomeContracts",
363
+ )
364
+ _require(
365
+ experiment.program_ref == run_request.program_ref
366
+ and experiment.public_context == run_request.public_context,
367
+ "the manifest and the run's request name a different improvement "
368
+ "program or public context",
369
+ )
370
+ _require(
371
+ evaluation_backend == run_request.evaluation_backend
372
+ and evaluation_backend == configuration.evaluation_backend,
373
+ "the evaluation backend this receipt would carry is not the one the "
374
+ "run's request and manifest were executed under",
375
+ )
376
+
377
+
378
+ def _require(condition: bool, message: str, **details: str) -> None:
379
+ """Raise a typed lineage refusal unless the condition holds."""
380
+ if condition:
381
+ return
382
+ reported: dict[str, JsonValue] = dict(details)
383
+ raise VerificationError(message, code=EPISODE_RECEIPT_INVALID, details=reported)
384
+
385
+
386
+ def _subject_trace(episode: NormalizedEpisode, variant: VariantName) -> NormalizedTrace:
387
+ """Return the one subject rollout this episode is allowed to carry."""
388
+ traces = [trace for trace in episode.traces if trace.agent_role == SUBJECT_AGENT]
389
+ if len(traces) == 1 and len(episode.traces) == 1:
390
+ return traces[0]
391
+ raise VerificationError(
392
+ f"episode {episode.episode_id} carries {len(episode.traces)} trace(s), "
393
+ f"{len(traces)} of them in the {SUBJECT_AGENT!r} seat; a v0.1 episode "
394
+ "carries exactly one subject trace and nothing else",
395
+ code=TRACE_ROLE_MISMATCH,
396
+ details={
397
+ "variant": variant.value,
398
+ "episode_id": episode.episode_id,
399
+ "traces": len(episode.traces),
400
+ "subject_traces": len(traces),
401
+ },
402
+ )
403
+
404
+
405
+ def _recorded_rewards(trace: NormalizedTrace) -> dict[str, float]:
406
+ """Copy every reward exactly as Verifiers scored it.
407
+
408
+ The score, not the weighted value. A weight is the Campaign's opinion about
409
+ how much a reward should count; the score is what the evaluation measured,
410
+ and a receipt records the measurement. The weight and the weighted value
411
+ both survive in the normalized episode the receipt's artifacts point at.
412
+ """
413
+ rewards: dict[str, float] = {}
414
+ for reward in trace.rewards:
415
+ _require_finite(reward.score, f"reward {reward.name!r}", trace)
416
+ rewards[reward.name] = reward.score
417
+ return rewards
418
+
419
+
420
+ def _recorded_metrics(trace: NormalizedTrace) -> dict[str, float | None]:
421
+ """Copy every metric exactly as recorded."""
422
+ for name, value in trace.metrics.items():
423
+ if value is not None:
424
+ _require_finite(value, f"metric {name!r}", trace, reward=False)
425
+ return dict(trace.metrics)
426
+
427
+
428
+ def _require_finite(
429
+ value: float, label: str, trace: NormalizedTrace, *, reward: bool = True
430
+ ) -> None:
431
+ """Refuse a number a receipt could not be digested with."""
432
+ if math.isfinite(value):
433
+ return
434
+ raise VerificationError(
435
+ f"the {label} recorded on trace {trace.trace_id} is not finite, so it "
436
+ "is not a measurement",
437
+ code=REWARD_NON_FINITE if reward else EVALUATION_OUTPUT_CORRUPT,
438
+ details={"trace_id": trace.trace_id, "value": repr(value)},
439
+ )
440
+
441
+
442
+ def _subject_runtime(trace: NormalizedTrace) -> SubjectRuntimeReceipt:
443
+ """Describe the box the subject actually executed in.
444
+
445
+ A Docker receipt must cite the image it ran, and it cites the content the
446
+ pinned reference names. Which platform that content was served on is a fact
447
+ about the machine rather than about the episode, so it lives in the
448
+ comparison's observed configuration and is left unset here.
449
+ """
450
+ return SubjectRuntimeReceipt(
451
+ kind="docker",
452
+ resolved_image_digest=validate_digest(trace.runtime.image_index_digest),
453
+ platform=None,
454
+ )
455
+
456
+
457
+ def _score_status(
458
+ episode: NormalizedEpisode, trace: NormalizedTrace, primary_reward: str
459
+ ) -> ScoreStatus:
460
+ """Say how much weight the recorded reward carries. Spec section 7.6."""
461
+ if not episode.ok or not trace.ok or episode.errors or trace.errors:
462
+ return ScoreStatus.ERRORED
463
+ if trace.reward(primary_reward) is None:
464
+ return ScoreStatus.MISSING
465
+ return ScoreStatus.VALID
466
+
467
+
468
+ def _evidence_status(
469
+ trace: NormalizedTrace, evidence: EvidenceRequirements
470
+ ) -> EvidenceStatus:
471
+ """Say how complete the supporting evidence is. Spec section 7.6.
472
+
473
+ Every artifact reference a receipt carries is a required field of the
474
+ execution result, hashed from the bytes on the way in, so their presence is
475
+ a property of the object rather than something to re-check. What is left to
476
+ decide is whether the subject configuration was recorded at all and whether
477
+ the Campaign asked for runtime evidence this release does not collect —
478
+ absence of a Relay is not incompleteness when the Campaign says runtime
479
+ evidence is not required.
480
+ """
481
+ configured = bool(trace.model_id and trace.harness_id and trace.tools)
482
+ if configured and evidence.runtime_evidence != "required":
483
+ return EvidenceStatus.COMPLETE
484
+ return EvidenceStatus.PARTIAL
485
+
486
+
487
+ def _variant_artifacts(result: VariantExecutionResult) -> list[ArtifactRef]:
488
+ """Reference the variant's evidence once, in one fixed order.
489
+
490
+ Spec section 7.6 permits a receipt to point at the variant's artifacts
491
+ rather than duplicating them per episode, and every one of these is a whole
492
+ file covering the variant: the configuration the engine resolved, the raw
493
+ upstream record, the engine's log, and the normalized projection this
494
+ receipt was built from.
495
+ """
496
+ return [
497
+ result.resolved_verifiers_config,
498
+ result.raw_traces,
499
+ result.eval_log,
500
+ result.normalized_episodes,
501
+ ]
502
+
503
+
504
+ def _receipt_id(*, run_id: str, variant: VariantName, episode_digest: Digest) -> str:
505
+ """Return a deterministic identifier in the shape :mod:`techtree.ids` issues.
506
+
507
+ Derived rather than random so that rebuilding a receipt from the same
508
+ evidence produces the same identifier, which is what lets a rebuild be
509
+ compared to what was stored.
510
+ """
511
+ digest = sha256_digest_bytes(
512
+ canonical_json_bytes(
513
+ {
514
+ "run_id": run_id,
515
+ "variant": variant.value,
516
+ "episode_digest": episode_digest,
517
+ }
518
+ )
519
+ )
520
+ _, _, hexadecimal = digest.partition(":")
521
+ return f"receipt_{hexadecimal[:_DERIVED_ID_LENGTH]}"
522
+
523
+
524
+ # ---------------------------------------------------------------------------
525
+ # One variant's receipts
526
+ # ---------------------------------------------------------------------------
527
+
528
+
529
+ def build_variant_receipts(
530
+ *,
531
+ run_request: RunRequest,
532
+ variant: VariantName,
533
+ experiment: ExperimentManifest,
534
+ result: VariantExecutionResult,
535
+ evaluation_backend: EvaluationBackendSpec,
536
+ ordered_task_hashes: Sequence[Digest],
537
+ primary_reward: str,
538
+ evidence: EvidenceRequirements,
539
+ ) -> list[EpisodeReceipt]:
540
+ """Build exactly one receipt per committed task, in committed order.
541
+
542
+ The join is on task hash and never on position in a file. Completion order
543
+ varies between two concurrent variants and has nothing to do with the
544
+ Campaign's commitment, so a receipt list built by walking the file would be
545
+ a list nobody could pair.
546
+
547
+ A variant with an unscored task is refused. One rollout that completed
548
+ cleanly and produced no primary reward means scoring did not run, and a
549
+ comparison computed over the tasks that happened to score would be a
550
+ comparison over a taskset the Campaign never committed to.
551
+
552
+ The lineage is settled before the episodes are looked at, so that an
553
+ execution filed under the wrong variant is reported as the provenance
554
+ failure it is rather than as whichever counting rule it happens to trip
555
+ first.
556
+ """
557
+ _require_lineage(
558
+ run_request=run_request,
559
+ variant=variant,
560
+ experiment=experiment,
561
+ raw_artifacts=result,
562
+ evaluation_backend=evaluation_backend,
563
+ )
564
+ committed = _committed_membership(ordered_task_hashes)
565
+ episodes = _episodes_by_task(result, committed, variant)
566
+
567
+ receipts = [
568
+ build_episode_receipt(
569
+ run_request=run_request,
570
+ variant=variant,
571
+ experiment=experiment,
572
+ episode=episodes[task_hash],
573
+ raw_artifacts=result,
574
+ evaluation_backend=evaluation_backend,
575
+ primary_reward=primary_reward,
576
+ evidence=evidence,
577
+ )
578
+ for task_hash in committed
579
+ ]
580
+ _require_every_task_scored(receipts, primary_reward, variant)
581
+ return receipts
582
+
583
+
584
+ def _committed_membership(ordered_task_hashes: Sequence[Digest]) -> list[Digest]:
585
+ """Revalidate the membership a variant's receipts are ordered by."""
586
+ committed = [validate_digest(value) for value in ordered_task_hashes]
587
+ if not committed:
588
+ raise VerificationError(
589
+ "a Campaign commits to at least one task, and this one commits to "
590
+ "none, so there is nothing to build receipts for",
591
+ code=TASK_MEMBERSHIP_MISMATCH,
592
+ details={"task_count": 0},
593
+ )
594
+ if len(set(committed)) != len(committed):
595
+ raise VerificationError(
596
+ "the committed membership names the same task twice, so a receipt "
597
+ "could be attributed to either position",
598
+ code=TASK_MEMBERSHIP_MISMATCH,
599
+ details={"task_count": len(committed)},
600
+ )
601
+ return committed
602
+
603
+
604
+ def _episodes_by_task(
605
+ result: VariantExecutionResult,
606
+ committed: Sequence[Digest],
607
+ variant: VariantName,
608
+ ) -> dict[Digest, NormalizedEpisode]:
609
+ """Index one variant's episodes by task, or say why they cannot be joined."""
610
+ if len(result.episodes) != len(committed):
611
+ raise VerificationError(
612
+ f"the {variant.value} variant recorded {len(result.episodes)} "
613
+ f"episodes for {len(committed)} committed tasks",
614
+ code=EPISODE_COUNT_MISMATCH,
615
+ details={
616
+ "variant": variant.value,
617
+ "recorded": len(result.episodes),
618
+ "committed": len(committed),
619
+ },
620
+ )
621
+
622
+ by_task: dict[Digest, NormalizedEpisode] = {}
623
+ for episode in result.episodes:
624
+ task_hash = validate_digest(episode.task_hash)
625
+ if task_hash in by_task:
626
+ raise VerificationError(
627
+ f"the {variant.value} variant scored task {task_hash} twice, so "
628
+ "one of the two results would have to be discarded",
629
+ code=TASK_MEMBERSHIP_MISMATCH,
630
+ details={"variant": variant.value, "task_hash": task_hash},
631
+ )
632
+ by_task[task_hash] = episode
633
+
634
+ missing: list[JsonValue] = [value for value in committed if value not in by_task]
635
+ unexpected: list[JsonValue] = [
636
+ value for value in sorted(set(by_task) - set(committed))
637
+ ]
638
+ if missing or unexpected:
639
+ raise VerificationError(
640
+ f"the {variant.value} variant scored a different set of tasks than "
641
+ "the Campaign commits to",
642
+ code=TASK_MEMBERSHIP_MISMATCH,
643
+ details={
644
+ "variant": variant.value,
645
+ "missing": missing,
646
+ "unexpected": unexpected,
647
+ },
648
+ )
649
+ return by_task
650
+
651
+
652
+ def _require_every_task_scored(
653
+ receipts: Sequence[EpisodeReceipt], primary_reward: str, variant: VariantName
654
+ ) -> None:
655
+ """Refuse a variant whose clean rollouts produced no primary reward."""
656
+ unscored: list[JsonValue] = [
657
+ receipt.task_hash
658
+ for receipt in receipts
659
+ if receipt.score_status is ScoreStatus.MISSING
660
+ ]
661
+ if not unscored:
662
+ return
663
+ raise VerificationError(
664
+ f"{len(unscored)} {variant.value} rollout(s) completed without scoring "
665
+ f"{primary_reward!r}, which is the reward this comparison is decided on",
666
+ code=REWARD_MISSING,
667
+ details={
668
+ "variant": variant.value,
669
+ "reward": primary_reward,
670
+ "task_hashes": unscored,
671
+ },
672
+ )