techtree 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. techtree/__init__.py +35 -0
  2. techtree/__main__.py +14 -0
  3. techtree/canonical.py +239 -0
  4. techtree/catalog/__init__.py +25 -0
  5. techtree/catalog/repository.py +400 -0
  6. techtree/catalog/service.py +419 -0
  7. techtree/cli/__init__.py +1 -0
  8. techtree/cli/app.py +416 -0
  9. techtree/cli/commands/__init__.py +1 -0
  10. techtree/cli/commands/climb.py +1223 -0
  11. techtree/cli/commands/doctor.py +147 -0
  12. techtree/cli/commands/engine.py +207 -0
  13. techtree/cli/commands/proof.py +556 -0
  14. techtree/cli/commands/publish.py +447 -0
  15. techtree/cli/commands/release.py +303 -0
  16. techtree/cli/commands/run.py +1067 -0
  17. techtree/cli/commands/setup.py +181 -0
  18. techtree/cli/commands/skill.py +221 -0
  19. techtree/cli/commands/uplift.py +698 -0
  20. techtree/cli/commands/withdraw.py +212 -0
  21. techtree/cli/confirm.py +47 -0
  22. techtree/cli/context.py +96 -0
  23. techtree/cli/invoke.py +220 -0
  24. techtree/cli/output.py +280 -0
  25. techtree/constants.py +138 -0
  26. techtree/crypto.py +128 -0
  27. techtree/doctor/__init__.py +1 -0
  28. techtree/doctor/checks.py +675 -0
  29. techtree/doctor/execution_checks.py +435 -0
  30. techtree/doctor/service.py +326 -0
  31. techtree/drafts/__init__.py +32 -0
  32. techtree/drafts/source.py +146 -0
  33. techtree/drafts/store.py +992 -0
  34. techtree/engines/__init__.py +1 -0
  35. techtree/engines/bundle.py +251 -0
  36. techtree/engines/installer.py +679 -0
  37. techtree/engines/registry.py +235 -0
  38. techtree/engines/runner.py +170 -0
  39. techtree/errors.py +262 -0
  40. techtree/fs.py +234 -0
  41. techtree/harness.py +108 -0
  42. techtree/identity/__init__.py +41 -0
  43. techtree/identity/models.py +113 -0
  44. techtree/identity/service.py +199 -0
  45. techtree/identity/store.py +263 -0
  46. techtree/ids.py +85 -0
  47. techtree/manifests/__init__.py +39 -0
  48. techtree/manifests/builder.py +433 -0
  49. techtree/manifests/compare.py +376 -0
  50. techtree/models/__init__.py +282 -0
  51. techtree/models/base.py +201 -0
  52. techtree/models/campaign.py +484 -0
  53. techtree/models/catalog.py +227 -0
  54. techtree/models/cli.py +151 -0
  55. techtree/models/climb.py +254 -0
  56. techtree/models/data_policy.py +130 -0
  57. techtree/models/engine.py +156 -0
  58. techtree/models/episode_receipt.py +130 -0
  59. techtree/models/evaluation_backend.py +113 -0
  60. techtree/models/experiment.py +154 -0
  61. techtree/models/run.py +214 -0
  62. techtree/models/skill.py +156 -0
  63. techtree/models/uplift_report.py +158 -0
  64. techtree/models/validation.py +299 -0
  65. techtree/paths.py +116 -0
  66. techtree/presentation/__init__.py +31 -0
  67. techtree/presentation/build.py +1242 -0
  68. techtree/presentation/compact.py +246 -0
  69. techtree/presentation/evidence.py +169 -0
  70. techtree/presentation/models.py +358 -0
  71. techtree/presentation/rich.py +312 -0
  72. techtree/presentation/sanitize.py +156 -0
  73. techtree/publication/__init__.py +44 -0
  74. techtree/publication/address.py +180 -0
  75. techtree/publication/coordinates.py +26 -0
  76. techtree/publication/journal.py +212 -0
  77. techtree/publication/keccak.py +183 -0
  78. techtree/publication/models.py +209 -0
  79. techtree/publication/offer.py +35 -0
  80. techtree/publication/service.py +618 -0
  81. techtree/publication/transport.py +296 -0
  82. techtree/publication/verify.py +242 -0
  83. techtree/publication/withdraw.py +156 -0
  84. techtree/py.typed +0 -0
  85. techtree/receipts/__init__.py +52 -0
  86. techtree/receipts/bundle.py +578 -0
  87. techtree/receipts/compare.py +1065 -0
  88. techtree/receipts/episode.py +672 -0
  89. techtree/receipts/execution.py +630 -0
  90. techtree/receipts/observed.py +474 -0
  91. techtree/receipts/set.py +336 -0
  92. techtree/receipts/uplift.py +655 -0
  93. techtree/receipts/verify.py +1055 -0
  94. techtree/release/__init__.py +9 -0
  95. techtree/release/bootstrap.py +509 -0
  96. techtree/release/checks.py +376 -0
  97. techtree/release/document.py +125 -0
  98. techtree/release/generate.py +221 -0
  99. techtree/release/models.py +293 -0
  100. techtree/release/provenance.py +109 -0
  101. techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
  102. techtree/resources/catalog/catalog.json +32 -0
  103. techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
  104. techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
  105. techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
  106. techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
  107. techtree/resources/engines/default/engine.json +20 -0
  108. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
  109. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
  110. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
  111. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
  112. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
  113. techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
  114. techtree/resources/engines/default/pyproject.toml +23 -0
  115. techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
  116. techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
  117. techtree/resources/engines/default/tools/normalize_validation.py +222 -0
  118. techtree/resources/engines/default/uv.lock +1758 -0
  119. techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
  120. techtree/resources/release/build-provenance.json +4 -0
  121. techtree/resources/release/release-core.json +24 -0
  122. techtree/runs/__init__.py +31 -0
  123. techtree/runs/artifacts.py +750 -0
  124. techtree/runs/child_registry.py +228 -0
  125. techtree/runs/events.py +478 -0
  126. techtree/runs/executor.py +140 -0
  127. techtree/runs/fake.py +741 -0
  128. techtree/runs/launcher.py +253 -0
  129. techtree/runs/machine.py +489 -0
  130. techtree/runs/real.py +789 -0
  131. techtree/runs/service.py +616 -0
  132. techtree/runs/store.py +555 -0
  133. techtree/runs/validation.py +259 -0
  134. techtree/runs/variants.py +684 -0
  135. techtree/settings.py +143 -0
  136. techtree/skills/__init__.py +14 -0
  137. techtree/skills/archive.py +282 -0
  138. techtree/skills/policy.py +62 -0
  139. techtree/skills/scanner.py +394 -0
  140. techtree/skills/service.py +752 -0
  141. techtree/skills/starter.py +434 -0
  142. techtree/tasksets/__init__.py +1 -0
  143. techtree/tasksets/membership.py +269 -0
  144. techtree/tasksets/provider.py +207 -0
  145. techtree/tasksets/resolver.py +311 -0
  146. techtree/tasksets/service.py +484 -0
  147. techtree/tasksets/verifiers_cli.py +538 -0
  148. techtree/uplift/__init__.py +20 -0
  149. techtree/uplift/context.py +544 -0
  150. techtree/uplift/derive.py +203 -0
  151. techtree/uplift/public_tasks.py +151 -0
  152. techtree/uplift/service.py +719 -0
  153. techtree/uplift/source.py +160 -0
  154. techtree/verifiers/__init__.py +31 -0
  155. techtree/verifiers/budget.py +219 -0
  156. techtree/verifiers/child.py +633 -0
  157. techtree/verifiers/compiler.py +432 -0
  158. techtree/verifiers/config.py +365 -0
  159. techtree/verifiers/credentials.py +321 -0
  160. techtree/verifiers/image.py +126 -0
  161. techtree/verifiers/models.py +527 -0
  162. techtree/verifiers/outputs.py +368 -0
  163. techtree/verifiers/progress.py +192 -0
  164. techtree/verifiers/supervisor.py +341 -0
  165. techtree/verifiers/verify.py +782 -0
  166. techtree/version.py +39 -0
  167. techtree/worker/__init__.py +18 -0
  168. techtree/worker/execute.py +487 -0
  169. techtree/worker/main.py +57 -0
  170. techtree-0.1.0.dist-info/METADATA +344 -0
  171. techtree-0.1.0.dist-info/RECORD +174 -0
  172. techtree-0.1.0.dist-info/WHEEL +4 -0
  173. techtree-0.1.0.dist-info/entry_points.txt +3 -0
  174. techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,719 @@
1
+ """The stage that closes a real run. Spec sections 7.9-7.10, 7.20.
2
+
3
+ The real executor stops the moment the evidence is complete: two variants
4
+ executed, their raw output retained, the engine's own normalized projection
5
+ written beside it, and a
6
+ :class:`~techtree.verifiers.models.RealExecutionResult` handed back. Spec
7
+ section 6.22 fixes that boundary deliberately — WP6 does not invent final
8
+ uplift — and until this module existed a real run therefore could not finish at
9
+ all. It is what turns the evidence into the run's result.
10
+
11
+ Everything it does is a pure function of files the run already owns, so it
12
+ spends nothing, starts nothing, and reaches nothing outside the run directory:
13
+
14
+ 1. ``building_receipts`` — one receipt per committed task per variant, written
15
+ into the run's receipt tree, signed under the local executor identity, and
16
+ committed to by an ordered receipt-set manifest per variant.
17
+ 2. ``verifying_comparison`` — one observed fingerprint per variant, taken from
18
+ the traces and the configuration the engine resolved, checked against the
19
+ manifests and against each other.
20
+ 3. ``building_report`` — the paired rewards, the aggregate, the report, its
21
+ signature, and the portable proof bundle the report is verified from before
22
+ it is written through the run store, so that the journal announces the
23
+ digest of a report whose proof already checked out.
24
+
25
+ Four choices are worth stating.
26
+
27
+ *The run's own copies are the only inputs.* The Campaign, the two manifests and
28
+ the committed membership come from ``inputs/``, which the artifact store
29
+ verified against the run's immutable request; the evidence comes from
30
+ ``verifiers/<variant>/run/``. Nothing is re-resolved from a draft, a catalog or
31
+ a settings file, so a run's result cannot change because something outside it
32
+ did.
33
+
34
+ *The receipts are written before the comparison is checked.* They are the
35
+ expensive evaluation's only durable scientific record, and a comparison that
36
+ fails is exactly the case where an operator most needs to read them.
37
+
38
+ *The proof is verified before the result is recorded.* Signing is not a
39
+ formality performed on the way out: the bundle is written, verified offline by
40
+ the same code a stranger would run, and only then does the report become the
41
+ run's result. A run whose proof does not check out fails with its evidence
42
+ intact rather than completing with a report nobody could verify.
43
+
44
+ *It records the run's completion itself.* The fake executor writes its own
45
+ result and appends its own completion event, and a real run needs the same two
46
+ acts performed by whoever built the report. Doing it here keeps
47
+ :func:`techtree.worker.execute.execute_run`'s contract — the report it verifies
48
+ against the journal is the report that was recorded — identical for both
49
+ executors.
50
+
51
+ Spec section 7.20's :class:`UpliftService` lives beside it, at the bottom of
52
+ this module. It is a different object with a different job: the report service
53
+ *closes* a run, and the uplift service is what the operator does with a run
54
+ that has already closed — export a sanitized context for a host agent to read,
55
+ hand over the verified text of the Skill that run measured, or prepare the
56
+ Skill-against-Skill comparison that follows from it. Nothing it does starts,
57
+ scores, or signs anything.
58
+ """
59
+
60
+ from __future__ import annotations
61
+
62
+ from collections.abc import Callable
63
+ from dataclasses import dataclass
64
+ from datetime import UTC, datetime
65
+ from pathlib import Path
66
+ from typing import Final
67
+
68
+ from pydantic import ValidationError as PydanticValidationError
69
+
70
+ from techtree.canonical import digest_object
71
+ from techtree.errors import PolicyError, ValidationError, VerificationError
72
+ from techtree.identity.models import ExecutorIdentity
73
+ from techtree.identity.service import IdentityService
74
+ from techtree.models.base import ObjectEnvelope
75
+ from techtree.models.episode_receipt import EpisodeReceipt
76
+ from techtree.models.experiment import ExperimentVariant
77
+ from techtree.models.run import RunPhase, RunRequest
78
+ from techtree.models.uplift_report import UpliftReport
79
+ from techtree.models.validation import TasksetLock
80
+ from techtree.paths import TechtreePaths
81
+ from techtree.receipts.bundle import (
82
+ PROOF_BUNDLE_INVALID,
83
+ LocalProofBundleContents,
84
+ ReferencedObject,
85
+ assess_local_attestation,
86
+ proof_bundle_dir,
87
+ write_local_bundle,
88
+ )
89
+ from techtree.receipts.compare import (
90
+ COMPARISON_INVALID,
91
+ ObservedVariant,
92
+ RealComparisonResult,
93
+ compare_real_variants,
94
+ observe_variant,
95
+ )
96
+ from techtree.receipts.episode import build_variant_receipts, experiment_variant_of
97
+ from techtree.receipts.execution import (
98
+ ComparisonExecutionRecord,
99
+ build_comparison_execution_record,
100
+ )
101
+ from techtree.receipts.observed import read_resolved_config
102
+ from techtree.receipts.set import (
103
+ ReceiptSetManifest,
104
+ build_receipt_set,
105
+ receipt_set_path,
106
+ write_receipt_set,
107
+ )
108
+ from techtree.receipts.uplift import (
109
+ aggregate_primary_result,
110
+ build_uplift_report,
111
+ pair_task_rewards,
112
+ summarize_receipts,
113
+ )
114
+ from techtree.receipts.verify import verify_local_bundle
115
+ from techtree.runs.artifacts import RunArtifactStore, RunInputBundle
116
+ from techtree.runs.events import DETAIL_RESULT_DIGEST, RUN_COMPLETED
117
+ from techtree.runs.executor import raise_if_cancel_requested
118
+ from techtree.runs.real import TASKSET_LOCK_FILENAME
119
+ from techtree.runs.service import RunService
120
+ from techtree.runs.store import RunStore
121
+ from techtree.skills.service import PreparedDraft, SkillPreparationService
122
+ from techtree.uplift.context import (
123
+ SkillImprovementContext,
124
+ build_improvement_context,
125
+ )
126
+ from techtree.uplift.public_tasks import public_projection_for
127
+ from techtree.uplift.source import VerifiedSourceSkill, read_verified_source_skill
128
+ from techtree.verifiers.models import (
129
+ RealExecutionResult,
130
+ RunPaths,
131
+ VariantExecutionResult,
132
+ VariantName,
133
+ )
134
+ from techtree.verifiers.outputs import RESOLVED_CONFIG_PATH
135
+
136
+ __all__ = [
137
+ "REAL_REPORT_STAGE_FAILED",
138
+ "SOURCE_RUN_NOT_USABLE",
139
+ "CompletedRun",
140
+ "RealUpliftReportService",
141
+ "UpliftService",
142
+ "VariantReceipts",
143
+ ]
144
+
145
+ #: Stable error code for an input this stage cannot read. The scientific
146
+ #: refusals report through spec section 15's own codes, raised by the modules
147
+ #: that own them.
148
+ REAL_REPORT_STAGE_FAILED: Final = "real_report_stage_failed"
149
+
150
+ #: A finished run that nothing may be derived from. Spec section 7.20's safety
151
+ #: rules all report through this one code, because they answer one question.
152
+ SOURCE_RUN_NOT_USABLE: Final = "source_run_not_usable"
153
+
154
+ #: Both sides, always in comparison order.
155
+ _VARIANT_ORDER: Final[tuple[VariantName, ...]] = (
156
+ VariantName.BASELINE,
157
+ VariantName.CANDIDATE,
158
+ )
159
+
160
+
161
+ @dataclass(frozen=True)
162
+ class CompletedRun:
163
+ """One finished run's signed result and the inputs it was executed from."""
164
+
165
+ report: UpliftReport
166
+ inputs: RunInputBundle
167
+
168
+
169
+ @dataclass(frozen=True)
170
+ class VariantReceipts:
171
+ """One variant's signed receipts, its commitment over them, and what it ran."""
172
+
173
+ receipts: list[EpisodeReceipt]
174
+ signed_receipts: list[ObjectEnvelope[EpisodeReceipt]]
175
+ receipt_set: ReceiptSetManifest
176
+ observed: ObservedVariant
177
+
178
+
179
+ class RealUpliftReportService:
180
+ """Completes a real run by turning its evidence into a signed UpliftReport."""
181
+
182
+ def __init__(
183
+ self,
184
+ *,
185
+ paths: TechtreePaths,
186
+ run_store: RunStore,
187
+ artifact_store: RunArtifactStore,
188
+ identity: IdentityService,
189
+ clock: Callable[[], datetime] | None = None,
190
+ ) -> None:
191
+ self._paths = paths
192
+ self._run_store = run_store
193
+ self._artifacts = artifact_store
194
+ self._identity = identity
195
+ self._clock = clock or _utc_now
196
+
197
+ def complete(
198
+ self, *, request: RunRequest, execution: RealExecutionResult
199
+ ) -> UpliftReport:
200
+ """Build, check, aggregate, sign, prove and record this run's result."""
201
+ run_id = request.run_id
202
+ raise_if_cancel_requested(self._run_store, run_id)
203
+
204
+ inputs = self._artifacts.load_inputs(run_id, request)
205
+ run_paths = RunPaths.for_run(self._paths, run_id)
206
+ lock = self._taskset_lock(run_paths, inputs)
207
+ # Before anything is sealed, so that a machine which cannot sign says so
208
+ # while its evidence is still the only thing at stake.
209
+ identity = self._identity.ensure()
210
+
211
+ self._run_store.append(run_id, phase=RunPhase.BUILDING_RECEIPTS)
212
+ sides = {
213
+ variant: self._build_variant(
214
+ request=request,
215
+ inputs=inputs,
216
+ run_paths=run_paths,
217
+ lock=lock,
218
+ variant=variant,
219
+ result=_side(execution, variant),
220
+ )
221
+ for variant in _VARIANT_ORDER
222
+ }
223
+ baseline = sides[VariantName.BASELINE]
224
+ candidate = sides[VariantName.CANDIDATE]
225
+
226
+ self._run_store.append(run_id, phase=RunPhase.VERIFYING_COMPARISON)
227
+ comparison = self._compare(
228
+ inputs=inputs, lock=lock, execution=execution, sides=sides
229
+ )
230
+
231
+ self._run_store.append(run_id, phase=RunPhase.BUILDING_REPORT)
232
+ report = self._report(
233
+ request=request,
234
+ inputs=inputs,
235
+ lock=lock,
236
+ identity=identity,
237
+ comparison=comparison,
238
+ baseline=baseline,
239
+ candidate=candidate,
240
+ )
241
+ # Decisions 0007 R6: what the comparison consumed, recorded beside
242
+ # what it measured and signed with the same key. Built from evidence
243
+ # the run already wrote, so it can neither change the report nor fail
244
+ # the run — a comparison whose economics are unknown is still a
245
+ # comparison, and this record is where that is said out loud.
246
+ execution_record = build_comparison_execution_record(
247
+ run_id=run_id,
248
+ campaign_spec_digest=request.campaign_spec_digest,
249
+ campaign_max_concurrent=inputs.campaign.execution.max_concurrent,
250
+ execution=execution,
251
+ run_root=self._paths.run_dir(run_id),
252
+ )
253
+ self._prove(
254
+ request=request,
255
+ inputs=inputs,
256
+ lock=lock,
257
+ identity=identity,
258
+ report=report,
259
+ baseline=baseline,
260
+ candidate=candidate,
261
+ execution_record=execution_record,
262
+ )
263
+
264
+ self._run_store.write_result(run_id, report)
265
+ self._run_store.append(
266
+ run_id,
267
+ phase=RunPhase.COMPLETED,
268
+ kind=RUN_COMPLETED,
269
+ details={DETAIL_RESULT_DIGEST: digest_object(report)},
270
+ )
271
+ return report
272
+
273
+ # -- receipts -----------------------------------------------------------
274
+
275
+ def _build_variant(
276
+ self,
277
+ *,
278
+ request: RunRequest,
279
+ inputs: RunInputBundle,
280
+ run_paths: RunPaths,
281
+ lock: TasksetLock,
282
+ variant: VariantName,
283
+ result: VariantExecutionResult,
284
+ ) -> VariantReceipts:
285
+ """Build, persist and commit to one variant's receipts."""
286
+ experiment = (
287
+ inputs.baseline if variant is VariantName.BASELINE else inputs.candidate
288
+ )
289
+ campaign = inputs.campaign
290
+ receipts = build_variant_receipts(
291
+ run_request=request,
292
+ variant=variant,
293
+ experiment=experiment,
294
+ result=result,
295
+ evaluation_backend=campaign.evaluation_backend,
296
+ ordered_task_hashes=lock.ordered_task_hashes,
297
+ primary_reward=campaign.scoring.primary_reward,
298
+ evidence=campaign.evidence,
299
+ )
300
+ for position, receipt in enumerate(receipts):
301
+ self._artifacts.write_episode_receipt(
302
+ request.run_id, position=position, receipt=receipt
303
+ )
304
+
305
+ # Signed here, at the moment the receipts exist and before anything is
306
+ # committed to them: the receipt-set manifest commits to the digest of
307
+ # each receipt's payload, which signing does not move, so the
308
+ # commitment and the signature are two independent seals over the same
309
+ # bytes rather than one wrapped in the other.
310
+ envelopes = [self._identity.sign_object(receipt) for receipt in receipts]
311
+ protocol_variant = experiment_variant_of(variant)
312
+ receipt_set = build_receipt_set(
313
+ run_id=request.run_id,
314
+ variant=protocol_variant,
315
+ experiment_manifest_digest=result.experiment_manifest_digest,
316
+ signed_receipts=envelopes,
317
+ ordered_task_hashes=lock.ordered_task_hashes,
318
+ )
319
+ write_receipt_set(
320
+ receipt_set, receipt_set_path(run_paths.root, protocol_variant)
321
+ )
322
+
323
+ return VariantReceipts(
324
+ receipts=receipts,
325
+ signed_receipts=envelopes,
326
+ receipt_set=receipt_set,
327
+ observed=observe_variant(
328
+ result=result,
329
+ resolved_config=read_resolved_config(
330
+ run_paths.variant_output_dir(variant) / RESOLVED_CONFIG_PATH
331
+ ),
332
+ runtime=campaign.subject.runtime,
333
+ ),
334
+ )
335
+
336
+ # -- comparison ---------------------------------------------------------
337
+
338
+ def _compare(
339
+ self,
340
+ *,
341
+ inputs: RunInputBundle,
342
+ lock: TasksetLock,
343
+ execution: RealExecutionResult,
344
+ sides: dict[VariantName, VariantReceipts],
345
+ ) -> RealComparisonResult:
346
+ """Check that the two executions were one experiment."""
347
+ return compare_real_variants(
348
+ campaign=inputs.campaign,
349
+ baseline_manifest=inputs.baseline,
350
+ candidate_manifest=inputs.candidate,
351
+ prepared_manifest_comparison=inputs.comparison,
352
+ baseline_receipts=sides[VariantName.BASELINE].receipts,
353
+ candidate_receipts=sides[VariantName.CANDIDATE].receipts,
354
+ taskset_lock=lock,
355
+ baseline_observed=sides[VariantName.BASELINE].observed,
356
+ candidate_observed=sides[VariantName.CANDIDATE].observed,
357
+ schedule=execution.schedule,
358
+ )
359
+
360
+ # -- the report ---------------------------------------------------------
361
+
362
+ def _report(
363
+ self,
364
+ *,
365
+ request: RunRequest,
366
+ inputs: RunInputBundle,
367
+ lock: TasksetLock,
368
+ identity: ExecutorIdentity,
369
+ comparison: RealComparisonResult,
370
+ baseline: VariantReceipts,
371
+ candidate: VariantReceipts,
372
+ ) -> UpliftReport:
373
+ """Aggregate the paired rewards and construct the report."""
374
+ campaign = inputs.campaign
375
+ deltas = pair_task_rewards(
376
+ baseline_receipts=baseline.receipts,
377
+ candidate_receipts=candidate.receipts,
378
+ ordered_task_hashes=comparison.ordered_task_hashes,
379
+ reward_name=campaign.scoring.primary_reward,
380
+ )
381
+ primary = aggregate_primary_result(deltas, campaign.scoring.primary_reward)
382
+ score, evidence = summarize_receipts(baseline.receipts, candidate.receipts)
383
+
384
+ return build_uplift_report(
385
+ run_request=request,
386
+ campaign=campaign,
387
+ # The run's own staged copy of the rights statement it executed
388
+ # under, which is what decides whether its report may be published.
389
+ data_policy=inputs.source.data_policy,
390
+ # The Campaign's own commitment. Every validation source in this
391
+ # build is required to have validated against exactly this receipt
392
+ # before a single episode is scored on the taskset, so it is the
393
+ # receipt this run was validated under and not merely a reference.
394
+ taskset_validation_receipt_digest=(
395
+ campaign.taskset.validation_receipt_digest
396
+ ),
397
+ baseline_manifest=inputs.baseline,
398
+ candidate_manifest=inputs.candidate,
399
+ baseline_receipt_set=baseline.receipt_set,
400
+ candidate_receipt_set=candidate.receipt_set,
401
+ comparison=comparison,
402
+ task_deltas=deltas,
403
+ primary=primary,
404
+ score=score,
405
+ evidence=evidence,
406
+ # The one argument that decides the grade, and it is decided by the
407
+ # decisions-0005 section 3.4 conditions rather than by the presence
408
+ # of a key. Every condition is checked; any failure withholds the
409
+ # verdict instead of claiming a grade this run did not earn.
410
+ attestation=assess_local_attestation(
411
+ identity=identity,
412
+ identity_self_check=self._identity.store.verify_pair(),
413
+ referenced_objects=self._referenced_objects(
414
+ request=request, inputs=inputs, lock=lock
415
+ ),
416
+ signed_receipts={
417
+ ExperimentVariant.BASELINE: baseline.signed_receipts,
418
+ ExperimentVariant.CANDIDATE: candidate.signed_receipts,
419
+ },
420
+ comparison=comparison.status,
421
+ score=score,
422
+ ).attestation,
423
+ created_at=self._clock(),
424
+ )
425
+
426
+ # -- the proof ----------------------------------------------------------
427
+
428
+ def _prove(
429
+ self,
430
+ *,
431
+ request: RunRequest,
432
+ inputs: RunInputBundle,
433
+ lock: TasksetLock,
434
+ identity: ExecutorIdentity,
435
+ report: UpliftReport,
436
+ baseline: VariantReceipts,
437
+ candidate: VariantReceipts,
438
+ execution_record: ComparisonExecutionRecord,
439
+ ) -> Path:
440
+ """Sign the report, write the portable proof, and verify it offline.
441
+
442
+ The bundle is verified before the report is recorded, from the bytes
443
+ just written, with the same verifier a stranger would use. That is what
444
+ turns "the report says P1" into "the conditions P1 names hold in this
445
+ directory": a bundle that does not verify fails the run, and no report
446
+ claiming a grade it cannot support becomes a result.
447
+ """
448
+ signed_report = self._identity.sign_object(report)
449
+ run_root = self._paths.run_dir(request.run_id)
450
+ directory = write_local_bundle(
451
+ run_root=run_root,
452
+ contents=LocalProofBundleContents(
453
+ identity=identity,
454
+ campaign=inputs.campaign,
455
+ data_policy=inputs.source.data_policy,
456
+ taskset_lock=lock,
457
+ validation_receipt=inputs.source.publisher_validation,
458
+ experiments={
459
+ ExperimentVariant.BASELINE: inputs.baseline,
460
+ ExperimentVariant.CANDIDATE: inputs.candidate,
461
+ },
462
+ receipt_sets={
463
+ ExperimentVariant.BASELINE: baseline.receipt_set,
464
+ ExperimentVariant.CANDIDATE: candidate.receipt_set,
465
+ },
466
+ receipts={
467
+ ExperimentVariant.BASELINE: baseline.signed_receipts,
468
+ ExperimentVariant.CANDIDATE: candidate.signed_receipts,
469
+ },
470
+ report=signed_report,
471
+ execution_record=self._identity.sign_object(execution_record),
472
+ ),
473
+ identity_service=self._identity,
474
+ )
475
+
476
+ verification = verify_local_bundle(directory)
477
+ if verification.verified:
478
+ return directory
479
+ raise VerificationError(
480
+ "this run's local proof does not verify from the bytes it just "
481
+ "wrote, so its report is not recorded as a result",
482
+ code=PROOF_BUNDLE_INVALID,
483
+ details={
484
+ "run_id": request.run_id,
485
+ "proof_grade": report.proof_grade,
486
+ "failed_checks": [message.id for message in verification.failures],
487
+ },
488
+ )
489
+
490
+ def _referenced_objects(
491
+ self, *, request: RunRequest, inputs: RunInputBundle, lock: TasksetLock
492
+ ) -> list[ReferencedObject]:
493
+ """Return every object this report cites, with the digest it cites it under.
494
+
495
+ Section 3.4's first condition is that all referenced artifact digests
496
+ verify, and this is the list: each object recomputed from itself and
497
+ compared against the digest some *other* document names it by, so no
498
+ value is ever checked against itself.
499
+ """
500
+ campaign = inputs.campaign
501
+ publisher_validation = inputs.source.publisher_validation
502
+ return [
503
+ ReferencedObject("campaign", campaign, request.campaign_spec_digest),
504
+ ReferencedObject(
505
+ "data-policy",
506
+ inputs.source.data_policy,
507
+ campaign.data_policy_digest,
508
+ ),
509
+ ReferencedObject(
510
+ "taskset-validation-receipt",
511
+ publisher_validation,
512
+ campaign.taskset.validation_receipt_digest,
513
+ ),
514
+ ReferencedObject(
515
+ "taskset-lock", lock, publisher_validation.taskset_lock_digest
516
+ ),
517
+ ReferencedObject(
518
+ "baseline-experiment",
519
+ inputs.baseline,
520
+ request.baseline_manifest_digest,
521
+ ),
522
+ ReferencedObject(
523
+ "candidate-experiment",
524
+ inputs.candidate,
525
+ request.candidate_manifest_digest,
526
+ ),
527
+ ]
528
+
529
+ # -- inputs -------------------------------------------------------------
530
+
531
+ def _taskset_lock(self, run_paths: RunPaths, inputs: RunInputBundle) -> TasksetLock:
532
+ """Read the run's own copy of the membership its episodes were joined on.
533
+
534
+ Written by the executor before anything was compiled, from the
535
+ validation this run was given, and read back here rather than re-derived
536
+ so that the receipts are ordered by the same list the normalizer used.
537
+ """
538
+ path = run_paths.inputs_dir / TASKSET_LOCK_FILENAME
539
+ try:
540
+ lock = TasksetLock.model_validate_json(path.read_bytes())
541
+ except OSError as error:
542
+ raise ValidationError(
543
+ "this run recorded no validated taskset lock, so its episodes "
544
+ "cannot be ordered by the membership they were scored under",
545
+ code=REAL_REPORT_STAGE_FAILED,
546
+ details={"run_id": inputs.request.run_id, "path": str(path)},
547
+ ) from error
548
+ except PydanticValidationError as error:
549
+ raise ValidationError(
550
+ f"this run's taskset lock cannot be read: {error.errors()[0]['msg']}",
551
+ code=REAL_REPORT_STAGE_FAILED,
552
+ details={"run_id": inputs.request.run_id, "path": str(path)},
553
+ ) from error
554
+
555
+ committed = list(inputs.campaign.taskset.membership.ordered_task_hashes)
556
+ if list(lock.ordered_task_hashes) != committed:
557
+ raise ValidationError(
558
+ "this run's taskset lock does not hold the tasks its Campaign "
559
+ "commits to",
560
+ code=COMPARISON_INVALID,
561
+ details={
562
+ "run_id": inputs.request.run_id,
563
+ "locked": len(lock.ordered_task_hashes),
564
+ "committed": len(committed),
565
+ },
566
+ )
567
+ return lock
568
+
569
+
570
+ def _side(
571
+ execution: RealExecutionResult, variant: VariantName
572
+ ) -> VariantExecutionResult:
573
+ """Return one variant's half of the execution result."""
574
+ return (
575
+ execution.baseline if variant is VariantName.BASELINE else execution.candidate
576
+ )
577
+
578
+
579
+ def _utc_now() -> datetime:
580
+ """Return the current instant in UTC."""
581
+ return datetime.now(UTC)
582
+
583
+
584
+ # ---------------------------------------------------------------------------
585
+ # Spec section 7.20: what an operator does with a run that has finished
586
+ # ---------------------------------------------------------------------------
587
+
588
+
589
+ class UpliftService:
590
+ """Exports improvement context and prepares Skill-replacement drafts.
591
+
592
+ Both operations read a run that already completed and neither changes it.
593
+ The safety rules in spec section 7.20 are all checked in one place,
594
+ :meth:`_completed_real_run`, because both operations rest on the same
595
+ claim: that this run really executed, really finished, and really verifies
596
+ from its own bytes. A context exported from a development-only run would
597
+ describe invented numbers, and a replacement prepared from one would
598
+ declare a baseline nothing measured.
599
+ """
600
+
601
+ def __init__(
602
+ self,
603
+ *,
604
+ paths: TechtreePaths,
605
+ run_service: RunService,
606
+ artifact_store: RunArtifactStore,
607
+ skill_service: SkillPreparationService,
608
+ ) -> None:
609
+ self._paths = paths
610
+ self._runs = run_service
611
+ self._artifacts = artifact_store
612
+ self._skills = skill_service
613
+
614
+ def improvement_context(self, run_id: str) -> SkillImprovementContext:
615
+ """Build sanitized local context from a completed real run."""
616
+ source = self._completed_real_run(run_id)
617
+ return build_improvement_context(
618
+ report=source.report,
619
+ candidate_receipts=self._artifacts.episode_receipts(
620
+ run_id, ExperimentVariant.CANDIDATE
621
+ ),
622
+ baseline_receipts=self._artifacts.episode_receipts(
623
+ run_id, ExperimentVariant.BASELINE
624
+ ),
625
+ campaign=source.inputs.campaign,
626
+ parent_skill=source.inputs.candidate_skill.artifact,
627
+ task_public_projection=public_projection_for(source.inputs.campaign),
628
+ )
629
+
630
+ def verified_source_skill(self, run_id: str) -> VerifiedSourceSkill:
631
+ """Return the verified text of the Skill this run measured.
632
+
633
+ The same run precondition the context rests on applies here, for the
634
+ same reason: text taken from a run that did not really execute would
635
+ be presented as the Skill a result was measured with, and no result
636
+ like that exists. What is returned is verified twice over — once when
637
+ the run's inputs are loaded, and again on the exact bytes handed back.
638
+ """
639
+ source = self._completed_real_run(run_id)
640
+ return read_verified_source_skill(source.inputs.candidate_skill, run_id=run_id)
641
+
642
+ def prepare_replacement(
643
+ self,
644
+ *,
645
+ source_run_id: str,
646
+ candidate_skill_path: Path,
647
+ candidate_label: str | None = None,
648
+ ) -> PreparedDraft:
649
+ """Prepare the run that compares the source run's Skill against a new one.
650
+
651
+ The baseline is the source run's *candidate* Skill, taken from that
652
+ run's own staged inputs — which the artifact store re-verifies file by
653
+ file against the artifact before handing them over — rather than
654
+ rescanned from wherever the participant originally wrote it. That is
655
+ what makes the second comparison's baseline the Skill the first
656
+ comparison actually measured, and not a directory that has moved on
657
+ since.
658
+ """
659
+ source = self._completed_real_run(source_run_id)
660
+ return self._skills.prepare_replacement(
661
+ source_campaign=source.inputs.campaign,
662
+ data_policy=source.inputs.source.data_policy,
663
+ publisher_validation=source.inputs.source.publisher_validation,
664
+ validation_evidence=source.inputs.validation_evidence,
665
+ source_report=source.report,
666
+ baseline_skill=source.inputs.candidate_skill,
667
+ candidate_skill_path=candidate_skill_path,
668
+ candidate_label=candidate_label,
669
+ )
670
+
671
+ # -- the one precondition both operations rest on -----------------------
672
+
673
+ def _completed_real_run(self, run_id: str) -> CompletedRun:
674
+ """Load a run that finished, executed for real, and verifies offline."""
675
+ # Raises with the run's own phase when it has not finished, and
676
+ # re-derives the report's digest against the journal that announced it.
677
+ report = self._runs.result(run_id)
678
+ inputs = self._artifacts.load_inputs(run_id, self._runs.request(run_id))
679
+
680
+ if report.proof_grade == "development_only":
681
+ raise PolicyError(
682
+ f"run {run_id} was executed by the development executor, so "
683
+ "its numbers are not evidence and nothing may be derived from "
684
+ "them",
685
+ code=SOURCE_RUN_NOT_USABLE,
686
+ details={"run_id": run_id, "proof_grade": report.proof_grade},
687
+ )
688
+
689
+ backends = {
690
+ receipt.execution_backend
691
+ for variant in ExperimentVariant
692
+ for receipt in self._artifacts.episode_receipts(run_id, variant)
693
+ }
694
+ if backends != {"verifiers"}:
695
+ raise PolicyError(
696
+ f"run {run_id} did not evaluate every episode for real, so it "
697
+ "is not a run another comparison can be built on",
698
+ code=SOURCE_RUN_NOT_USABLE,
699
+ details={
700
+ "run_id": run_id,
701
+ "execution_backends": ", ".join(sorted(backends)),
702
+ },
703
+ )
704
+
705
+ verification = verify_local_bundle(
706
+ proof_bundle_dir(self._paths.run_dir(run_id))
707
+ )
708
+ if not verification.verified:
709
+ raise VerificationError(
710
+ f"run {run_id}'s local proof does not verify, so nothing may "
711
+ "be derived from what it reported",
712
+ code=PROOF_BUNDLE_INVALID,
713
+ details={
714
+ "run_id": run_id,
715
+ "failed_checks": [message.id for message in verification.failures],
716
+ },
717
+ )
718
+
719
+ return CompletedRun(report=report, inputs=inputs)