techtree 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. techtree/__init__.py +35 -0
  2. techtree/__main__.py +14 -0
  3. techtree/canonical.py +239 -0
  4. techtree/catalog/__init__.py +25 -0
  5. techtree/catalog/repository.py +400 -0
  6. techtree/catalog/service.py +419 -0
  7. techtree/cli/__init__.py +1 -0
  8. techtree/cli/app.py +416 -0
  9. techtree/cli/commands/__init__.py +1 -0
  10. techtree/cli/commands/climb.py +1223 -0
  11. techtree/cli/commands/doctor.py +147 -0
  12. techtree/cli/commands/engine.py +207 -0
  13. techtree/cli/commands/proof.py +556 -0
  14. techtree/cli/commands/publish.py +447 -0
  15. techtree/cli/commands/release.py +303 -0
  16. techtree/cli/commands/run.py +1067 -0
  17. techtree/cli/commands/setup.py +181 -0
  18. techtree/cli/commands/skill.py +221 -0
  19. techtree/cli/commands/uplift.py +698 -0
  20. techtree/cli/commands/withdraw.py +212 -0
  21. techtree/cli/confirm.py +47 -0
  22. techtree/cli/context.py +96 -0
  23. techtree/cli/invoke.py +220 -0
  24. techtree/cli/output.py +280 -0
  25. techtree/constants.py +138 -0
  26. techtree/crypto.py +128 -0
  27. techtree/doctor/__init__.py +1 -0
  28. techtree/doctor/checks.py +675 -0
  29. techtree/doctor/execution_checks.py +435 -0
  30. techtree/doctor/service.py +326 -0
  31. techtree/drafts/__init__.py +32 -0
  32. techtree/drafts/source.py +146 -0
  33. techtree/drafts/store.py +992 -0
  34. techtree/engines/__init__.py +1 -0
  35. techtree/engines/bundle.py +251 -0
  36. techtree/engines/installer.py +679 -0
  37. techtree/engines/registry.py +235 -0
  38. techtree/engines/runner.py +170 -0
  39. techtree/errors.py +262 -0
  40. techtree/fs.py +234 -0
  41. techtree/harness.py +108 -0
  42. techtree/identity/__init__.py +41 -0
  43. techtree/identity/models.py +113 -0
  44. techtree/identity/service.py +199 -0
  45. techtree/identity/store.py +263 -0
  46. techtree/ids.py +85 -0
  47. techtree/manifests/__init__.py +39 -0
  48. techtree/manifests/builder.py +433 -0
  49. techtree/manifests/compare.py +376 -0
  50. techtree/models/__init__.py +282 -0
  51. techtree/models/base.py +201 -0
  52. techtree/models/campaign.py +484 -0
  53. techtree/models/catalog.py +227 -0
  54. techtree/models/cli.py +151 -0
  55. techtree/models/climb.py +254 -0
  56. techtree/models/data_policy.py +130 -0
  57. techtree/models/engine.py +156 -0
  58. techtree/models/episode_receipt.py +130 -0
  59. techtree/models/evaluation_backend.py +113 -0
  60. techtree/models/experiment.py +154 -0
  61. techtree/models/run.py +214 -0
  62. techtree/models/skill.py +156 -0
  63. techtree/models/uplift_report.py +158 -0
  64. techtree/models/validation.py +299 -0
  65. techtree/paths.py +116 -0
  66. techtree/presentation/__init__.py +31 -0
  67. techtree/presentation/build.py +1242 -0
  68. techtree/presentation/compact.py +246 -0
  69. techtree/presentation/evidence.py +169 -0
  70. techtree/presentation/models.py +358 -0
  71. techtree/presentation/rich.py +312 -0
  72. techtree/presentation/sanitize.py +156 -0
  73. techtree/publication/__init__.py +44 -0
  74. techtree/publication/address.py +180 -0
  75. techtree/publication/coordinates.py +26 -0
  76. techtree/publication/journal.py +212 -0
  77. techtree/publication/keccak.py +183 -0
  78. techtree/publication/models.py +209 -0
  79. techtree/publication/offer.py +35 -0
  80. techtree/publication/service.py +618 -0
  81. techtree/publication/transport.py +296 -0
  82. techtree/publication/verify.py +242 -0
  83. techtree/publication/withdraw.py +156 -0
  84. techtree/py.typed +0 -0
  85. techtree/receipts/__init__.py +52 -0
  86. techtree/receipts/bundle.py +578 -0
  87. techtree/receipts/compare.py +1065 -0
  88. techtree/receipts/episode.py +672 -0
  89. techtree/receipts/execution.py +630 -0
  90. techtree/receipts/observed.py +474 -0
  91. techtree/receipts/set.py +336 -0
  92. techtree/receipts/uplift.py +655 -0
  93. techtree/receipts/verify.py +1055 -0
  94. techtree/release/__init__.py +9 -0
  95. techtree/release/bootstrap.py +509 -0
  96. techtree/release/checks.py +376 -0
  97. techtree/release/document.py +125 -0
  98. techtree/release/generate.py +221 -0
  99. techtree/release/models.py +293 -0
  100. techtree/release/provenance.py +109 -0
  101. techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
  102. techtree/resources/catalog/catalog.json +32 -0
  103. techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
  104. techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
  105. techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
  106. techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
  107. techtree/resources/engines/default/engine.json +20 -0
  108. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
  109. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
  110. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
  111. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
  112. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
  113. techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
  114. techtree/resources/engines/default/pyproject.toml +23 -0
  115. techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
  116. techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
  117. techtree/resources/engines/default/tools/normalize_validation.py +222 -0
  118. techtree/resources/engines/default/uv.lock +1758 -0
  119. techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
  120. techtree/resources/release/build-provenance.json +4 -0
  121. techtree/resources/release/release-core.json +24 -0
  122. techtree/runs/__init__.py +31 -0
  123. techtree/runs/artifacts.py +750 -0
  124. techtree/runs/child_registry.py +228 -0
  125. techtree/runs/events.py +478 -0
  126. techtree/runs/executor.py +140 -0
  127. techtree/runs/fake.py +741 -0
  128. techtree/runs/launcher.py +253 -0
  129. techtree/runs/machine.py +489 -0
  130. techtree/runs/real.py +789 -0
  131. techtree/runs/service.py +616 -0
  132. techtree/runs/store.py +555 -0
  133. techtree/runs/validation.py +259 -0
  134. techtree/runs/variants.py +684 -0
  135. techtree/settings.py +143 -0
  136. techtree/skills/__init__.py +14 -0
  137. techtree/skills/archive.py +282 -0
  138. techtree/skills/policy.py +62 -0
  139. techtree/skills/scanner.py +394 -0
  140. techtree/skills/service.py +752 -0
  141. techtree/skills/starter.py +434 -0
  142. techtree/tasksets/__init__.py +1 -0
  143. techtree/tasksets/membership.py +269 -0
  144. techtree/tasksets/provider.py +207 -0
  145. techtree/tasksets/resolver.py +311 -0
  146. techtree/tasksets/service.py +484 -0
  147. techtree/tasksets/verifiers_cli.py +538 -0
  148. techtree/uplift/__init__.py +20 -0
  149. techtree/uplift/context.py +544 -0
  150. techtree/uplift/derive.py +203 -0
  151. techtree/uplift/public_tasks.py +151 -0
  152. techtree/uplift/service.py +719 -0
  153. techtree/uplift/source.py +160 -0
  154. techtree/verifiers/__init__.py +31 -0
  155. techtree/verifiers/budget.py +219 -0
  156. techtree/verifiers/child.py +633 -0
  157. techtree/verifiers/compiler.py +432 -0
  158. techtree/verifiers/config.py +365 -0
  159. techtree/verifiers/credentials.py +321 -0
  160. techtree/verifiers/image.py +126 -0
  161. techtree/verifiers/models.py +527 -0
  162. techtree/verifiers/outputs.py +368 -0
  163. techtree/verifiers/progress.py +192 -0
  164. techtree/verifiers/supervisor.py +341 -0
  165. techtree/verifiers/verify.py +782 -0
  166. techtree/version.py +39 -0
  167. techtree/worker/__init__.py +18 -0
  168. techtree/worker/execute.py +487 -0
  169. techtree/worker/main.py +57 -0
  170. techtree-0.1.0.dist-info/METADATA +344 -0
  171. techtree-0.1.0.dist-info/RECORD +174 -0
  172. techtree-0.1.0.dist-info/WHEEL +4 -0
  173. techtree-0.1.0.dist-info/entry_points.txt +3 -0
  174. techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,655 @@
1
+ """Reward aggregation and the real report. Spec section 7.10.
2
+
3
+ Nothing here scores anything. Every number in an
4
+ :class:`~techtree.models.uplift_report.UpliftReport` is arithmetic over rewards
5
+ Verifiers recorded and Techtree copied into receipts without touching them, and
6
+ the arithmetic is the smallest that answers the Campaign's own question: two
7
+ means, their difference, and how many tasks moved which way.
8
+
9
+ Five rules shape it, and each one is a way of not lying.
10
+
11
+ *The join is on task identity.* Two variants complete their episodes in whatever
12
+ order the provider and the containers produced. Pairing by position in a list
13
+ would compare task 3 against task 7 in exactly the runs where concurrency
14
+ worked, so the pairing is by task hash and the row order is the TasksetLock's.
15
+
16
+ *One receipt per task per variant, or nothing.* A missing task is not a shorter
17
+ list and a duplicated task is not a tie-break; both are refusals, because a mean
18
+ over the tasks that happened to arrive is a mean over a taskset nobody committed
19
+ to.
20
+
21
+ *The score, not the weighted value.* A weight is the Campaign's opinion about
22
+ how much a reward should count. The comparison is on the reward Verifiers
23
+ scored, consistently on both sides, exactly as the receipts hold it.
24
+
25
+ *A relative improvement over nothing is not a number.* When the baseline mean is
26
+ zero, ``relative_delta`` is null. Reporting zero or infinity would each be a
27
+ different false statement, and the recorded evidence this was built against has
28
+ a zero baseline, so it is the ordinary case rather than the edge one.
29
+
30
+ *A tie is exact equality.* Spec section 7.10 permits a Campaign-declared
31
+ tolerance instead, and the frozen
32
+ :class:`~techtree.models.campaign.ScoringSpec` declares none, so there is no
33
+ tolerance to apply. For the discrete rewards v0.1 measures — ``exact_match`` is
34
+ 0.0 or 1.0 — exact equality is also the right rule rather than a fallback.
35
+
36
+ WHAT GRADE A REAL REPORT CARRIES, AND WHY
37
+
38
+ Decisions document 0005 section 3.4 lets a report claim ``proof_grade: P1``
39
+ only when its receipts and the report itself are wrapped in *signed* envelopes
40
+ under the local executor identity, that identity's public key travels with
41
+ them, the comparison is controlled, and the score is valid. Whether those
42
+ conditions hold is not this module's judgement to make: it arrives as
43
+ :class:`LocalAttestation`, decided by
44
+ :func:`techtree.receipts.bundle.assess_local_attestation`, which checks each
45
+ condition by name and re-checks them all against the written bundle before the
46
+ report is recorded.
47
+
48
+ What this module owns is the consequence. The frozen model offers exactly two
49
+ grades and couples the weaker one to the verdict: a ``development_only`` report
50
+ must reach a ``development_only`` decision and must not be publication
51
+ eligible. So an unattested real report states everything it measured —
52
+ execution completed, score valid, evidence complete, comparison controlled,
53
+ both means, every task delta — and withholds the *verdict*, because the verdict
54
+ is the field the frozen model ties to the proof grade. It is still not a fake
55
+ report: a fake one is ``development_only`` in its score, evidence and
56
+ comparison statuses too, and this one is not.
57
+
58
+ An attested one carries P1 and the verdict :func:`decide_uplift` computed:
59
+ accepted, rejected or inconclusive, by the Campaign's own predeclared rules.
60
+ """
61
+
62
+ from __future__ import annotations
63
+
64
+ import math
65
+ from collections.abc import Iterable, Sequence
66
+ from datetime import datetime
67
+ from enum import StrEnum
68
+ from typing import Literal
69
+
70
+ from techtree.canonical import digest_object
71
+ from techtree.constants import UPLIFT_SCHEMA_VERSION
72
+ from techtree.errors import VerificationError
73
+ from techtree.ids import new_id
74
+ from techtree.models.base import Digest, JsonValue
75
+ from techtree.models.campaign import SUBJECT_AGENT, CampaignSpec
76
+ from techtree.models.data_policy import DataPolicy
77
+ from techtree.models.episode_receipt import (
78
+ EpisodeReceipt,
79
+ EvidenceStatus,
80
+ ScoreStatus,
81
+ )
82
+ from techtree.models.experiment import ExperimentManifest, ExperimentVariant
83
+ from techtree.models.run import RunRequest
84
+ from techtree.models.uplift_report import (
85
+ ComparisonStatus,
86
+ ExecutionStatus,
87
+ PrimaryUpliftResult,
88
+ PublicationStatus,
89
+ TaskDelta,
90
+ UpliftDecision,
91
+ UpliftReport,
92
+ UpliftStatuses,
93
+ )
94
+ from techtree.receipts.compare import COMPARISON_INVALID, RealComparisonResult
95
+ from techtree.receipts.episode import (
96
+ REWARD_MISSING,
97
+ REWARD_NON_FINITE,
98
+ TASK_MEMBERSHIP_MISMATCH,
99
+ )
100
+ from techtree.receipts.set import ReceiptSetManifest
101
+ from techtree.tasksets.membership import membership_digest
102
+
103
+ __all__ = [
104
+ "LocalAttestation",
105
+ "aggregate_primary_result",
106
+ "build_uplift_report",
107
+ "decide_uplift",
108
+ "pair_task_rewards",
109
+ "proof_grade_for",
110
+ "publication_eligible_for",
111
+ "publication_status_for",
112
+ "summarize_receipts",
113
+ ]
114
+
115
+
116
+ class LocalAttestation(StrEnum):
117
+ """Whether the local executor identity binds this report's evidence.
118
+
119
+ Spelled as an argument rather than inferred, so that the one condition
120
+ separating a P1 report from an ungraded one is visible at the call site
121
+ that knows the answer.
122
+ """
123
+
124
+ #: At least one decisions-0005 section 3.4 condition does not hold, so
125
+ #: nothing has sealed this evidence in the sense the grade requires.
126
+ UNATTESTED = "unattested"
127
+
128
+ #: Every receipt and the report travel in signed envelopes under the
129
+ #: participant's own Ed25519 key, the public key travels with them, and
130
+ #: every other section 3.4 condition holds.
131
+ LOCAL_ED25519 = "local_ed25519"
132
+
133
+
134
+ # ---------------------------------------------------------------------------
135
+ # Pairing
136
+ # ---------------------------------------------------------------------------
137
+
138
+
139
+ def pair_task_rewards(
140
+ *,
141
+ baseline_receipts: Sequence[EpisodeReceipt],
142
+ candidate_receipts: Sequence[EpisodeReceipt],
143
+ ordered_task_hashes: Sequence[Digest],
144
+ reward_name: str,
145
+ ) -> list[TaskDelta]:
146
+ """Join the two variants by task hash and return rows in TasksetLock order."""
147
+ committed = list(ordered_task_hashes)
148
+ if not committed:
149
+ raise VerificationError(
150
+ "a comparison covers at least one committed task, and this one covers none",
151
+ code=TASK_MEMBERSHIP_MISMATCH,
152
+ details={"task_count": 0},
153
+ )
154
+ if len(set(committed)) != len(committed):
155
+ raise VerificationError(
156
+ "the committed membership names the same task twice, so a pair "
157
+ "could be built from either of two receipts",
158
+ code=TASK_MEMBERSHIP_MISMATCH,
159
+ details={"task_count": len(committed)},
160
+ )
161
+
162
+ baseline = _rewards_by_task(baseline_receipts, reward_name, committed, "baseline")
163
+ candidate = _rewards_by_task(
164
+ candidate_receipts, reward_name, committed, "candidate"
165
+ )
166
+
167
+ return [
168
+ TaskDelta(
169
+ task_hash=task_hash,
170
+ baseline_reward=baseline[task_hash],
171
+ candidate_reward=candidate[task_hash],
172
+ delta=candidate[task_hash] - baseline[task_hash],
173
+ )
174
+ for task_hash in committed
175
+ ]
176
+
177
+
178
+ def _rewards_by_task(
179
+ receipts: Sequence[EpisodeReceipt],
180
+ reward_name: str,
181
+ committed: Sequence[Digest],
182
+ label: str,
183
+ ) -> dict[Digest, float]:
184
+ """Read one variant's primary reward per task, refusing anything ambiguous."""
185
+ rewards: dict[Digest, float] = {}
186
+ for receipt in receipts:
187
+ traces = receipt.named_traces.get(SUBJECT_AGENT, [])
188
+ if len(traces) != 1:
189
+ raise VerificationError(
190
+ f"a {label} receipt carries {len(traces)} subject traces; one "
191
+ "episode has exactly one",
192
+ code=TASK_MEMBERSHIP_MISMATCH,
193
+ details={"task_hash": receipt.task_hash, "traces": len(traces)},
194
+ )
195
+ if receipt.task_hash in rewards:
196
+ raise VerificationError(
197
+ f"the {label} variant scored task {receipt.task_hash} twice, so "
198
+ "one of the two rewards would have to be discarded",
199
+ code=TASK_MEMBERSHIP_MISMATCH,
200
+ details={"variant": label, "task_hash": receipt.task_hash},
201
+ )
202
+ reward = traces[0].rewards.get(reward_name)
203
+ if reward is None:
204
+ raise VerificationError(
205
+ f"a {label} receipt records no {reward_name!r} reward, which is "
206
+ "the reward this comparison is decided on",
207
+ code=REWARD_MISSING,
208
+ details={"task_hash": receipt.task_hash, "reward": reward_name},
209
+ )
210
+ _require_finite(reward, label, receipt.task_hash)
211
+ rewards[receipt.task_hash] = reward
212
+
213
+ missing: list[JsonValue] = [value for value in committed if value not in rewards]
214
+ unexpected: list[JsonValue] = [
215
+ value for value in sorted(set(rewards) - set(committed))
216
+ ]
217
+ if missing or unexpected:
218
+ raise VerificationError(
219
+ f"the {label} variant scored a different set of tasks than the "
220
+ "Campaign commits to",
221
+ code=TASK_MEMBERSHIP_MISMATCH,
222
+ details={"variant": label, "missing": missing, "unexpected": unexpected},
223
+ )
224
+ return rewards
225
+
226
+
227
+ def _require_finite(value: float, label: str, task_hash: Digest) -> None:
228
+ """Refuse a reward that cannot be averaged or canonically written down."""
229
+ if math.isfinite(value):
230
+ return
231
+ raise VerificationError(
232
+ f"the {label} reward recorded for task {task_hash} is not finite, so it "
233
+ "is not a measurement",
234
+ code=REWARD_NON_FINITE,
235
+ details={"variant": label, "task_hash": task_hash, "value": repr(value)},
236
+ )
237
+
238
+
239
+ # ---------------------------------------------------------------------------
240
+ # Aggregation
241
+ # ---------------------------------------------------------------------------
242
+
243
+
244
+ def aggregate_primary_result(
245
+ deltas: Sequence[TaskDelta], reward_name: str
246
+ ) -> PrimaryUpliftResult:
247
+ """Compute the headline result from the paired rows and nothing else."""
248
+ if not deltas:
249
+ raise VerificationError(
250
+ "an uplift result summarizes at least one paired task",
251
+ code=TASK_MEMBERSHIP_MISMATCH,
252
+ details={"reward": reward_name, "task_count": 0},
253
+ )
254
+ for delta in deltas:
255
+ _require_finite(delta.baseline_reward, "baseline", delta.task_hash)
256
+ _require_finite(delta.candidate_reward, "candidate", delta.task_hash)
257
+
258
+ baseline_mean = _mean(delta.baseline_reward for delta in deltas)
259
+ candidate_mean = _mean(delta.candidate_reward for delta in deltas)
260
+ absolute = candidate_mean - baseline_mean
261
+ for value, label in (
262
+ (baseline_mean, "baseline mean"),
263
+ (candidate_mean, "candidate mean"),
264
+ (absolute, "absolute delta"),
265
+ ):
266
+ if not math.isfinite(value):
267
+ raise VerificationError(
268
+ f"the {label} over these rewards is not a finite number",
269
+ code=REWARD_NON_FINITE,
270
+ details={"reward": reward_name, "task_count": len(deltas)},
271
+ )
272
+
273
+ return PrimaryUpliftResult(
274
+ reward_name=reward_name,
275
+ baseline_mean=baseline_mean,
276
+ candidate_mean=candidate_mean,
277
+ absolute_delta=absolute,
278
+ # Section 7.10: null over a zero baseline. Any number here would be an
279
+ # invented one.
280
+ relative_delta=None if baseline_mean == 0.0 else absolute / baseline_mean,
281
+ wins=sum(
282
+ 1 for delta in deltas if delta.candidate_reward > delta.baseline_reward
283
+ ),
284
+ losses=sum(
285
+ 1 for delta in deltas if delta.candidate_reward < delta.baseline_reward
286
+ ),
287
+ ties=sum(
288
+ 1 for delta in deltas if delta.candidate_reward == delta.baseline_reward
289
+ ),
290
+ )
291
+
292
+
293
+ def _mean(values: Iterable[float]) -> float:
294
+ collected = list(values)
295
+ return sum(collected) / len(collected)
296
+
297
+
298
+ # ---------------------------------------------------------------------------
299
+ # The verdict
300
+ # ---------------------------------------------------------------------------
301
+
302
+
303
+ def decide_uplift(
304
+ *,
305
+ campaign: CampaignSpec,
306
+ comparison: RealComparisonResult,
307
+ primary: PrimaryUpliftResult,
308
+ ) -> UpliftDecision:
309
+ """Apply the Campaign's own acceptance rules, and no others.
310
+
311
+ ``inconclusive`` is reached when the Campaign predeclared no rule that can
312
+ decide: a scoring contract that neither requires the candidate to out-score
313
+ the baseline nor sets a minimum delta would accept a regression, so calling
314
+ such a result "accepted" would report a verdict nobody specified.
315
+ """
316
+ if not comparison.controlled:
317
+ return UpliftDecision.INVALID
318
+
319
+ scoring = campaign.scoring
320
+ if not scoring.require_candidate_above_baseline and (
321
+ scoring.minimum_absolute_delta == 0.0
322
+ ):
323
+ return UpliftDecision.INCONCLUSIVE
324
+ if scoring.require_candidate_above_baseline and primary.absolute_delta <= 0.0:
325
+ return UpliftDecision.REJECTED
326
+ if primary.absolute_delta < scoring.minimum_absolute_delta:
327
+ return UpliftDecision.REJECTED
328
+ return UpliftDecision.ACCEPTED
329
+
330
+
331
+ def summarize_receipts(
332
+ baseline_receipts: Sequence[EpisodeReceipt],
333
+ candidate_receipts: Sequence[EpisodeReceipt],
334
+ ) -> tuple[ScoreStatus, EvidenceStatus]:
335
+ """Return the score and evidence statuses the whole comparison carries.
336
+
337
+ A comparison is only as good as its weakest receipt. One rollout whose
338
+ scoring errored makes the aggregate score invalid rather than making the
339
+ other rollouts' scores worth less, and one receipt whose evidence is
340
+ partial makes the comparison's evidence partial.
341
+ """
342
+ receipts = [*baseline_receipts, *candidate_receipts]
343
+ if not receipts:
344
+ return ScoreStatus.MISSING, EvidenceStatus.NOT_COLLECTED
345
+
346
+ score = (
347
+ ScoreStatus.VALID
348
+ if all(receipt.score_status is ScoreStatus.VALID for receipt in receipts)
349
+ else ScoreStatus.INVALID
350
+ )
351
+ evidence = (
352
+ EvidenceStatus.COMPLETE
353
+ if all(
354
+ receipt.evidence_status is EvidenceStatus.COMPLETE for receipt in receipts
355
+ )
356
+ else EvidenceStatus.PARTIAL
357
+ )
358
+ return score, evidence
359
+
360
+
361
+ def publication_status_for(data_policy: DataPolicy) -> PublicationStatus:
362
+ """Return where a fresh report stands with respect to being published.
363
+
364
+ Two of the five statuses can be true of a report that has just been
365
+ written, and which one it is comes from the rights statement the run
366
+ executed under rather than from anything the run did. A DataPolicy that
367
+ does not make the uplift report public is a policy under which publishing
368
+ it is not a thing anybody may choose: that is ``blocked``, decided once,
369
+ before anybody is offered anything. A policy that does make it public
370
+ leaves the choice open, and a choice nobody has made yet is
371
+ ``not_requested``.
372
+
373
+ The other three describe an attempt rather than a report. A run's
374
+ publication journal owns those, because it is the only record of an
375
+ attempt; nothing rewrites a signed report to say it was published, and the
376
+ proof it belongs to is what makes that impossible as well as wrong.
377
+ """
378
+ if data_policy.derived_artifacts.uplift_report == "public":
379
+ return PublicationStatus.NOT_REQUESTED
380
+ return PublicationStatus.BLOCKED
381
+
382
+
383
+ def publication_eligible_for(
384
+ *,
385
+ grade: Literal["development_only", "P1"],
386
+ publication: PublicationStatus,
387
+ ) -> bool:
388
+ """Return whether this report may be published at all.
389
+
390
+ Until there was somewhere to publish to, this was a constant ``False``
391
+ whose comment said why: no route, no credential, no server. There is a
392
+ route now, so the flag has to answer the question it is named for, and
393
+ the answer has two halves and no third.
394
+
395
+ *The evidence has to be worth publishing.* A ``development_only`` report is
396
+ either invented numbers or a real comparison nothing sealed, and neither is
397
+ evidence of anything. Only a P1 report — signed receipts, a signed report,
398
+ the public key travelling with them, a controlled comparison and a valid
399
+ score — is eligible, which is the same bar decisions 0005 section 3.4 sets
400
+ for the grade itself.
401
+
402
+ *The rights have to permit it.* A report whose policy blocks publication is
403
+ not eligible however good the evidence is.
404
+
405
+ Both halves are also what the frozen model's own validator insists on, so
406
+ the computation cannot produce a report the model would refuse.
407
+ """
408
+ return grade == "P1" and publication is not PublicationStatus.BLOCKED
409
+
410
+
411
+ def proof_grade_for(
412
+ *,
413
+ attestation: LocalAttestation,
414
+ comparison: ComparisonStatus,
415
+ score: ScoreStatus,
416
+ ) -> Literal["development_only", "P1"]:
417
+ """Return the strongest grade this report is entitled to claim.
418
+
419
+ Decisions document 0005 section 3.4. The signature conditions are the
420
+ caller's to establish and are summarized by ``attestation``; the two
421
+ conditions this function can check itself are checked here, so a signed
422
+ report over an uncontrolled comparison still cannot claim P1.
423
+ """
424
+ controlled = comparison in (
425
+ ComparisonStatus.CONTROLLED,
426
+ ComparisonStatus.CONTROLLED_WITH_WARNINGS,
427
+ )
428
+ if (
429
+ attestation is LocalAttestation.LOCAL_ED25519
430
+ and controlled
431
+ and score is ScoreStatus.VALID
432
+ ):
433
+ return "P1"
434
+ return "development_only"
435
+
436
+
437
+ # ---------------------------------------------------------------------------
438
+ # The report
439
+ # ---------------------------------------------------------------------------
440
+
441
+
442
+ def build_uplift_report(
443
+ *,
444
+ run_request: RunRequest,
445
+ campaign: CampaignSpec,
446
+ data_policy: DataPolicy,
447
+ taskset_validation_receipt_digest: Digest,
448
+ baseline_manifest: ExperimentManifest,
449
+ candidate_manifest: ExperimentManifest,
450
+ baseline_receipt_set: ReceiptSetManifest,
451
+ candidate_receipt_set: ReceiptSetManifest,
452
+ comparison: RealComparisonResult,
453
+ task_deltas: Sequence[TaskDelta],
454
+ primary: PrimaryUpliftResult,
455
+ score: ScoreStatus,
456
+ evidence: EvidenceStatus,
457
+ attestation: LocalAttestation,
458
+ created_at: datetime,
459
+ ) -> UpliftReport:
460
+ """Construct the canonical real local report, or refuse to construct one.
461
+
462
+ Two conditions are refusals rather than statuses. A comparison that is not
463
+ controlled did not measure the Skill, and a score that is not valid did not
464
+ measure anything; in both cases spec section 7.10's decision is ``invalid``,
465
+ and the frozen model has no way to carry that verdict without also claiming
466
+ the P1 grade that decisions document 0005 forbids an uncontrolled
467
+ comparison. A report saying "invalid" is worth less than a run that failed
468
+ with the reason, so the reason is raised.
469
+
470
+ ``score`` and ``evidence`` are passed in rather than derived from the
471
+ receipt sets, which commit to receipts by digest and hold no statuses;
472
+ :func:`summarize_receipts` computes them from the receipts themselves.
473
+
474
+ The DataPolicy is passed in whole rather than by digest because publication
475
+ eligibility is read off its terms. It is checked against the digest the
476
+ run's request names, so a report cannot cite a Campaign's rights statement
477
+ and be graded under a different one.
478
+ """
479
+ _require_reportable(comparison, score, run_request)
480
+ _require_lineage(
481
+ run_request=run_request,
482
+ campaign=campaign,
483
+ data_policy=data_policy,
484
+ baseline_manifest=baseline_manifest,
485
+ candidate_manifest=candidate_manifest,
486
+ baseline_receipt_set=baseline_receipt_set,
487
+ candidate_receipt_set=candidate_receipt_set,
488
+ comparison=comparison,
489
+ )
490
+
491
+ grade = proof_grade_for(
492
+ attestation=attestation, comparison=comparison.status, score=score
493
+ )
494
+ publication = publication_status_for(data_policy)
495
+ decision = (
496
+ decide_uplift(campaign=campaign, comparison=comparison, primary=primary)
497
+ if grade == "P1"
498
+ # An unsigned real report withholds the verdict rather than presenting
499
+ # one the frozen model would have to grade P1. See the module docstring.
500
+ else UpliftDecision.DEVELOPMENT_ONLY
501
+ )
502
+
503
+ return UpliftReport(
504
+ schema_version=UPLIFT_SCHEMA_VERSION,
505
+ id=new_id("uplift"),
506
+ run_id=run_request.run_id,
507
+ campaign_spec_digest=run_request.campaign_spec_digest,
508
+ program_ref=run_request.program_ref,
509
+ public_context=run_request.public_context,
510
+ data_policy_digest=run_request.data_policy_digest,
511
+ outcome_contract_digest=run_request.outcome_contract_digest,
512
+ evaluation_backend=campaign.evaluation_backend,
513
+ taskset_validation_receipt_digest=taskset_validation_receipt_digest,
514
+ baseline_manifest_digest=run_request.baseline_manifest_digest,
515
+ candidate_manifest_digest=run_request.candidate_manifest_digest,
516
+ statuses=UpliftStatuses(
517
+ # A report is only built for an execution that finished both
518
+ # variants; a partial one fails the run instead (spec 6.17).
519
+ execution=ExecutionStatus.COMPLETED,
520
+ score=score,
521
+ evidence=evidence,
522
+ comparison=comparison.status,
523
+ # Nothing has been uploaded: a report is written before anybody has
524
+ # been asked whether to publish it, so the only two answers
525
+ # available here are "nobody has asked" and "the rights forbid it".
526
+ # Spec section 7.10.
527
+ publication=publication,
528
+ ),
529
+ manifest_comparison=comparison.manifest_comparison,
530
+ primary_result=primary,
531
+ task_deltas=list(task_deltas),
532
+ decision=decision,
533
+ proof_grade=grade,
534
+ publication_eligible=publication_eligible_for(
535
+ grade=grade, publication=publication
536
+ ),
537
+ created_at=created_at,
538
+ )
539
+
540
+
541
+ def _require_reportable(
542
+ comparison: RealComparisonResult, score: ScoreStatus, run_request: RunRequest
543
+ ) -> None:
544
+ """Refuse to write a report over evidence that decided nothing."""
545
+ if not comparison.controlled:
546
+ raise VerificationError(
547
+ "this run's two variants were not one controlled experiment, so "
548
+ "there is no uplift to report: "
549
+ + "; ".join(check.detail for check in comparison.failures),
550
+ code=COMPARISON_INVALID,
551
+ details={
552
+ "run_id": run_request.run_id,
553
+ "comparison": comparison.status.value,
554
+ "failed_checks": [check.id for check in comparison.failures],
555
+ },
556
+ )
557
+ if score is not ScoreStatus.VALID:
558
+ raise VerificationError(
559
+ f"this run's recorded scores are {score.value}, so the comparison "
560
+ "measured nothing that may be reported",
561
+ code=COMPARISON_INVALID,
562
+ details={"run_id": run_request.run_id, "score": score.value},
563
+ )
564
+
565
+
566
+ def _require_lineage(
567
+ *,
568
+ run_request: RunRequest,
569
+ campaign: CampaignSpec,
570
+ data_policy: DataPolicy,
571
+ baseline_manifest: ExperimentManifest,
572
+ candidate_manifest: ExperimentManifest,
573
+ baseline_receipt_set: ReceiptSetManifest,
574
+ candidate_receipt_set: ReceiptSetManifest,
575
+ comparison: RealComparisonResult,
576
+ ) -> None:
577
+ """Require every object the report cites to belong to the same run.
578
+
579
+ The receipt sets are inputs to this check rather than fields of the report:
580
+ the frozen report has nowhere to carry a receipt-set digest, and what it can
581
+ still do is refuse to summarize receipts that belong to a different run,
582
+ variant or manifest.
583
+ """
584
+ campaign_digest = digest_object(campaign)
585
+ for label, expected, found in (
586
+ ("Campaign", run_request.campaign_spec_digest, campaign_digest),
587
+ (
588
+ "baseline manifest",
589
+ run_request.baseline_manifest_digest,
590
+ digest_object(baseline_manifest),
591
+ ),
592
+ (
593
+ "candidate manifest",
594
+ run_request.candidate_manifest_digest,
595
+ digest_object(candidate_manifest),
596
+ ),
597
+ (
598
+ "DataPolicy",
599
+ run_request.data_policy_digest,
600
+ campaign.data_policy_digest,
601
+ ),
602
+ # The document itself, and not only the digest the Campaign names.
603
+ # Publication eligibility is read off these terms, so the terms have to
604
+ # be the ones this run was executed under.
605
+ (
606
+ "DataPolicy document",
607
+ run_request.data_policy_digest,
608
+ digest_object(data_policy),
609
+ ),
610
+ ):
611
+ if expected != found:
612
+ raise VerificationError(
613
+ f"the {label} this report would cite is not the one the run's "
614
+ "request names",
615
+ code=COMPARISON_INVALID,
616
+ details={
617
+ "run_id": run_request.run_id,
618
+ "expected": expected,
619
+ "found": found,
620
+ },
621
+ )
622
+
623
+ committed = list(comparison.ordered_task_hashes)
624
+ for manifest_set, variant, manifest_digest in (
625
+ (
626
+ baseline_receipt_set,
627
+ ExperimentVariant.BASELINE,
628
+ run_request.baseline_manifest_digest,
629
+ ),
630
+ (
631
+ candidate_receipt_set,
632
+ ExperimentVariant.CANDIDATE,
633
+ run_request.candidate_manifest_digest,
634
+ ),
635
+ ):
636
+ if (
637
+ manifest_set.run_id == run_request.run_id
638
+ and manifest_set.variant is variant
639
+ and manifest_set.experiment_manifest_digest == manifest_digest
640
+ and manifest_set.receipt_count == len(committed)
641
+ and manifest_set.task_membership_digest == membership_digest(committed)
642
+ ):
643
+ continue
644
+ raise VerificationError(
645
+ f"the {variant.value} receipt set does not commit to this run's "
646
+ f"{variant.value} receipts",
647
+ code=COMPARISON_INVALID,
648
+ details={
649
+ "run_id": run_request.run_id,
650
+ "variant": variant.value,
651
+ "receipt_set_run_id": manifest_set.run_id,
652
+ "receipt_count": manifest_set.receipt_count,
653
+ "committed": len(committed),
654
+ },
655
+ )