techtree 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- techtree/__init__.py +35 -0
- techtree/__main__.py +14 -0
- techtree/canonical.py +239 -0
- techtree/catalog/__init__.py +25 -0
- techtree/catalog/repository.py +400 -0
- techtree/catalog/service.py +419 -0
- techtree/cli/__init__.py +1 -0
- techtree/cli/app.py +416 -0
- techtree/cli/commands/__init__.py +1 -0
- techtree/cli/commands/climb.py +1223 -0
- techtree/cli/commands/doctor.py +147 -0
- techtree/cli/commands/engine.py +207 -0
- techtree/cli/commands/proof.py +556 -0
- techtree/cli/commands/publish.py +447 -0
- techtree/cli/commands/release.py +303 -0
- techtree/cli/commands/run.py +1067 -0
- techtree/cli/commands/setup.py +181 -0
- techtree/cli/commands/skill.py +221 -0
- techtree/cli/commands/uplift.py +698 -0
- techtree/cli/commands/withdraw.py +212 -0
- techtree/cli/confirm.py +47 -0
- techtree/cli/context.py +96 -0
- techtree/cli/invoke.py +220 -0
- techtree/cli/output.py +280 -0
- techtree/constants.py +138 -0
- techtree/crypto.py +128 -0
- techtree/doctor/__init__.py +1 -0
- techtree/doctor/checks.py +675 -0
- techtree/doctor/execution_checks.py +435 -0
- techtree/doctor/service.py +326 -0
- techtree/drafts/__init__.py +32 -0
- techtree/drafts/source.py +146 -0
- techtree/drafts/store.py +992 -0
- techtree/engines/__init__.py +1 -0
- techtree/engines/bundle.py +251 -0
- techtree/engines/installer.py +679 -0
- techtree/engines/registry.py +235 -0
- techtree/engines/runner.py +170 -0
- techtree/errors.py +262 -0
- techtree/fs.py +234 -0
- techtree/harness.py +108 -0
- techtree/identity/__init__.py +41 -0
- techtree/identity/models.py +113 -0
- techtree/identity/service.py +199 -0
- techtree/identity/store.py +263 -0
- techtree/ids.py +85 -0
- techtree/manifests/__init__.py +39 -0
- techtree/manifests/builder.py +433 -0
- techtree/manifests/compare.py +376 -0
- techtree/models/__init__.py +282 -0
- techtree/models/base.py +201 -0
- techtree/models/campaign.py +484 -0
- techtree/models/catalog.py +227 -0
- techtree/models/cli.py +151 -0
- techtree/models/climb.py +254 -0
- techtree/models/data_policy.py +130 -0
- techtree/models/engine.py +156 -0
- techtree/models/episode_receipt.py +130 -0
- techtree/models/evaluation_backend.py +113 -0
- techtree/models/experiment.py +154 -0
- techtree/models/run.py +214 -0
- techtree/models/skill.py +156 -0
- techtree/models/uplift_report.py +158 -0
- techtree/models/validation.py +299 -0
- techtree/paths.py +116 -0
- techtree/presentation/__init__.py +31 -0
- techtree/presentation/build.py +1242 -0
- techtree/presentation/compact.py +246 -0
- techtree/presentation/evidence.py +169 -0
- techtree/presentation/models.py +358 -0
- techtree/presentation/rich.py +312 -0
- techtree/presentation/sanitize.py +156 -0
- techtree/publication/__init__.py +44 -0
- techtree/publication/address.py +180 -0
- techtree/publication/coordinates.py +26 -0
- techtree/publication/journal.py +212 -0
- techtree/publication/keccak.py +183 -0
- techtree/publication/models.py +209 -0
- techtree/publication/offer.py +35 -0
- techtree/publication/service.py +618 -0
- techtree/publication/transport.py +296 -0
- techtree/publication/verify.py +242 -0
- techtree/publication/withdraw.py +156 -0
- techtree/py.typed +0 -0
- techtree/receipts/__init__.py +52 -0
- techtree/receipts/bundle.py +578 -0
- techtree/receipts/compare.py +1065 -0
- techtree/receipts/episode.py +672 -0
- techtree/receipts/execution.py +630 -0
- techtree/receipts/observed.py +474 -0
- techtree/receipts/set.py +336 -0
- techtree/receipts/uplift.py +655 -0
- techtree/receipts/verify.py +1055 -0
- techtree/release/__init__.py +9 -0
- techtree/release/bootstrap.py +509 -0
- techtree/release/checks.py +376 -0
- techtree/release/document.py +125 -0
- techtree/release/generate.py +221 -0
- techtree/release/models.py +293 -0
- techtree/release/provenance.py +109 -0
- techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
- techtree/resources/catalog/catalog.json +32 -0
- techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
- techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
- techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
- techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
- techtree/resources/engines/default/engine.json +20 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
- techtree/resources/engines/default/pyproject.toml +23 -0
- techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
- techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
- techtree/resources/engines/default/tools/normalize_validation.py +222 -0
- techtree/resources/engines/default/uv.lock +1758 -0
- techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
- techtree/resources/release/build-provenance.json +4 -0
- techtree/resources/release/release-core.json +24 -0
- techtree/runs/__init__.py +31 -0
- techtree/runs/artifacts.py +750 -0
- techtree/runs/child_registry.py +228 -0
- techtree/runs/events.py +478 -0
- techtree/runs/executor.py +140 -0
- techtree/runs/fake.py +741 -0
- techtree/runs/launcher.py +253 -0
- techtree/runs/machine.py +489 -0
- techtree/runs/real.py +789 -0
- techtree/runs/service.py +616 -0
- techtree/runs/store.py +555 -0
- techtree/runs/validation.py +259 -0
- techtree/runs/variants.py +684 -0
- techtree/settings.py +143 -0
- techtree/skills/__init__.py +14 -0
- techtree/skills/archive.py +282 -0
- techtree/skills/policy.py +62 -0
- techtree/skills/scanner.py +394 -0
- techtree/skills/service.py +752 -0
- techtree/skills/starter.py +434 -0
- techtree/tasksets/__init__.py +1 -0
- techtree/tasksets/membership.py +269 -0
- techtree/tasksets/provider.py +207 -0
- techtree/tasksets/resolver.py +311 -0
- techtree/tasksets/service.py +484 -0
- techtree/tasksets/verifiers_cli.py +538 -0
- techtree/uplift/__init__.py +20 -0
- techtree/uplift/context.py +544 -0
- techtree/uplift/derive.py +203 -0
- techtree/uplift/public_tasks.py +151 -0
- techtree/uplift/service.py +719 -0
- techtree/uplift/source.py +160 -0
- techtree/verifiers/__init__.py +31 -0
- techtree/verifiers/budget.py +219 -0
- techtree/verifiers/child.py +633 -0
- techtree/verifiers/compiler.py +432 -0
- techtree/verifiers/config.py +365 -0
- techtree/verifiers/credentials.py +321 -0
- techtree/verifiers/image.py +126 -0
- techtree/verifiers/models.py +527 -0
- techtree/verifiers/outputs.py +368 -0
- techtree/verifiers/progress.py +192 -0
- techtree/verifiers/supervisor.py +341 -0
- techtree/verifiers/verify.py +782 -0
- techtree/version.py +39 -0
- techtree/worker/__init__.py +18 -0
- techtree/worker/execute.py +487 -0
- techtree/worker/main.py +57 -0
- techtree-0.1.0.dist-info/METADATA +344 -0
- techtree-0.1.0.dist-info/RECORD +174 -0
- techtree-0.1.0.dist-info/WHEEL +4 -0
- techtree-0.1.0.dist-info/entry_points.txt +3 -0
- techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,655 @@
|
|
|
1
|
+
"""Reward aggregation and the real report. Spec section 7.10.
|
|
2
|
+
|
|
3
|
+
Nothing here scores anything. Every number in an
|
|
4
|
+
:class:`~techtree.models.uplift_report.UpliftReport` is arithmetic over rewards
|
|
5
|
+
Verifiers recorded and Techtree copied into receipts without touching them, and
|
|
6
|
+
the arithmetic is the smallest that answers the Campaign's own question: two
|
|
7
|
+
means, their difference, and how many tasks moved which way.
|
|
8
|
+
|
|
9
|
+
Five rules shape it, and each one is a way of not lying.
|
|
10
|
+
|
|
11
|
+
*The join is on task identity.* Two variants complete their episodes in whatever
|
|
12
|
+
order the provider and the containers produced. Pairing by position in a list
|
|
13
|
+
would compare task 3 against task 7 in exactly the runs where concurrency
|
|
14
|
+
worked, so the pairing is by task hash and the row order is the TasksetLock's.
|
|
15
|
+
|
|
16
|
+
*One receipt per task per variant, or nothing.* A missing task is not a shorter
|
|
17
|
+
list and a duplicated task is not a tie-break; both are refusals, because a mean
|
|
18
|
+
over the tasks that happened to arrive is a mean over a taskset nobody committed
|
|
19
|
+
to.
|
|
20
|
+
|
|
21
|
+
*The score, not the weighted value.* A weight is the Campaign's opinion about
|
|
22
|
+
how much a reward should count. The comparison is on the reward Verifiers
|
|
23
|
+
scored, consistently on both sides, exactly as the receipts hold it.
|
|
24
|
+
|
|
25
|
+
*A relative improvement over nothing is not a number.* When the baseline mean is
|
|
26
|
+
zero, ``relative_delta`` is null. Reporting zero or infinity would each be a
|
|
27
|
+
different false statement, and the recorded evidence this was built against has
|
|
28
|
+
a zero baseline, so it is the ordinary case rather than the edge one.
|
|
29
|
+
|
|
30
|
+
*A tie is exact equality.* Spec section 7.10 permits a Campaign-declared
|
|
31
|
+
tolerance instead, and the frozen
|
|
32
|
+
:class:`~techtree.models.campaign.ScoringSpec` declares none, so there is no
|
|
33
|
+
tolerance to apply. For the discrete rewards v0.1 measures — ``exact_match`` is
|
|
34
|
+
0.0 or 1.0 — exact equality is also the right rule rather than a fallback.
|
|
35
|
+
|
|
36
|
+
WHAT GRADE A REAL REPORT CARRIES, AND WHY
|
|
37
|
+
|
|
38
|
+
Decisions document 0005 section 3.4 lets a report claim ``proof_grade: P1``
|
|
39
|
+
only when its receipts and the report itself are wrapped in *signed* envelopes
|
|
40
|
+
under the local executor identity, that identity's public key travels with
|
|
41
|
+
them, the comparison is controlled, and the score is valid. Whether those
|
|
42
|
+
conditions hold is not this module's judgement to make: it arrives as
|
|
43
|
+
:class:`LocalAttestation`, decided by
|
|
44
|
+
:func:`techtree.receipts.bundle.assess_local_attestation`, which checks each
|
|
45
|
+
condition by name and re-checks them all against the written bundle before the
|
|
46
|
+
report is recorded.
|
|
47
|
+
|
|
48
|
+
What this module owns is the consequence. The frozen model offers exactly two
|
|
49
|
+
grades and couples the weaker one to the verdict: a ``development_only`` report
|
|
50
|
+
must reach a ``development_only`` decision and must not be publication
|
|
51
|
+
eligible. So an unattested real report states everything it measured —
|
|
52
|
+
execution completed, score valid, evidence complete, comparison controlled,
|
|
53
|
+
both means, every task delta — and withholds the *verdict*, because the verdict
|
|
54
|
+
is the field the frozen model ties to the proof grade. It is still not a fake
|
|
55
|
+
report: a fake one is ``development_only`` in its score, evidence and
|
|
56
|
+
comparison statuses too, and this one is not.
|
|
57
|
+
|
|
58
|
+
An attested one carries P1 and the verdict :func:`decide_uplift` computed:
|
|
59
|
+
accepted, rejected or inconclusive, by the Campaign's own predeclared rules.
|
|
60
|
+
"""
|
|
61
|
+
|
|
62
|
+
from __future__ import annotations
|
|
63
|
+
|
|
64
|
+
import math
|
|
65
|
+
from collections.abc import Iterable, Sequence
|
|
66
|
+
from datetime import datetime
|
|
67
|
+
from enum import StrEnum
|
|
68
|
+
from typing import Literal
|
|
69
|
+
|
|
70
|
+
from techtree.canonical import digest_object
|
|
71
|
+
from techtree.constants import UPLIFT_SCHEMA_VERSION
|
|
72
|
+
from techtree.errors import VerificationError
|
|
73
|
+
from techtree.ids import new_id
|
|
74
|
+
from techtree.models.base import Digest, JsonValue
|
|
75
|
+
from techtree.models.campaign import SUBJECT_AGENT, CampaignSpec
|
|
76
|
+
from techtree.models.data_policy import DataPolicy
|
|
77
|
+
from techtree.models.episode_receipt import (
|
|
78
|
+
EpisodeReceipt,
|
|
79
|
+
EvidenceStatus,
|
|
80
|
+
ScoreStatus,
|
|
81
|
+
)
|
|
82
|
+
from techtree.models.experiment import ExperimentManifest, ExperimentVariant
|
|
83
|
+
from techtree.models.run import RunRequest
|
|
84
|
+
from techtree.models.uplift_report import (
|
|
85
|
+
ComparisonStatus,
|
|
86
|
+
ExecutionStatus,
|
|
87
|
+
PrimaryUpliftResult,
|
|
88
|
+
PublicationStatus,
|
|
89
|
+
TaskDelta,
|
|
90
|
+
UpliftDecision,
|
|
91
|
+
UpliftReport,
|
|
92
|
+
UpliftStatuses,
|
|
93
|
+
)
|
|
94
|
+
from techtree.receipts.compare import COMPARISON_INVALID, RealComparisonResult
|
|
95
|
+
from techtree.receipts.episode import (
|
|
96
|
+
REWARD_MISSING,
|
|
97
|
+
REWARD_NON_FINITE,
|
|
98
|
+
TASK_MEMBERSHIP_MISMATCH,
|
|
99
|
+
)
|
|
100
|
+
from techtree.receipts.set import ReceiptSetManifest
|
|
101
|
+
from techtree.tasksets.membership import membership_digest
|
|
102
|
+
|
|
103
|
+
__all__ = [
|
|
104
|
+
"LocalAttestation",
|
|
105
|
+
"aggregate_primary_result",
|
|
106
|
+
"build_uplift_report",
|
|
107
|
+
"decide_uplift",
|
|
108
|
+
"pair_task_rewards",
|
|
109
|
+
"proof_grade_for",
|
|
110
|
+
"publication_eligible_for",
|
|
111
|
+
"publication_status_for",
|
|
112
|
+
"summarize_receipts",
|
|
113
|
+
]
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
class LocalAttestation(StrEnum):
|
|
117
|
+
"""Whether the local executor identity binds this report's evidence.
|
|
118
|
+
|
|
119
|
+
Spelled as an argument rather than inferred, so that the one condition
|
|
120
|
+
separating a P1 report from an ungraded one is visible at the call site
|
|
121
|
+
that knows the answer.
|
|
122
|
+
"""
|
|
123
|
+
|
|
124
|
+
#: At least one decisions-0005 section 3.4 condition does not hold, so
|
|
125
|
+
#: nothing has sealed this evidence in the sense the grade requires.
|
|
126
|
+
UNATTESTED = "unattested"
|
|
127
|
+
|
|
128
|
+
#: Every receipt and the report travel in signed envelopes under the
|
|
129
|
+
#: participant's own Ed25519 key, the public key travels with them, and
|
|
130
|
+
#: every other section 3.4 condition holds.
|
|
131
|
+
LOCAL_ED25519 = "local_ed25519"
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
# ---------------------------------------------------------------------------
|
|
135
|
+
# Pairing
|
|
136
|
+
# ---------------------------------------------------------------------------
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def pair_task_rewards(
|
|
140
|
+
*,
|
|
141
|
+
baseline_receipts: Sequence[EpisodeReceipt],
|
|
142
|
+
candidate_receipts: Sequence[EpisodeReceipt],
|
|
143
|
+
ordered_task_hashes: Sequence[Digest],
|
|
144
|
+
reward_name: str,
|
|
145
|
+
) -> list[TaskDelta]:
|
|
146
|
+
"""Join the two variants by task hash and return rows in TasksetLock order."""
|
|
147
|
+
committed = list(ordered_task_hashes)
|
|
148
|
+
if not committed:
|
|
149
|
+
raise VerificationError(
|
|
150
|
+
"a comparison covers at least one committed task, and this one covers none",
|
|
151
|
+
code=TASK_MEMBERSHIP_MISMATCH,
|
|
152
|
+
details={"task_count": 0},
|
|
153
|
+
)
|
|
154
|
+
if len(set(committed)) != len(committed):
|
|
155
|
+
raise VerificationError(
|
|
156
|
+
"the committed membership names the same task twice, so a pair "
|
|
157
|
+
"could be built from either of two receipts",
|
|
158
|
+
code=TASK_MEMBERSHIP_MISMATCH,
|
|
159
|
+
details={"task_count": len(committed)},
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
baseline = _rewards_by_task(baseline_receipts, reward_name, committed, "baseline")
|
|
163
|
+
candidate = _rewards_by_task(
|
|
164
|
+
candidate_receipts, reward_name, committed, "candidate"
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
return [
|
|
168
|
+
TaskDelta(
|
|
169
|
+
task_hash=task_hash,
|
|
170
|
+
baseline_reward=baseline[task_hash],
|
|
171
|
+
candidate_reward=candidate[task_hash],
|
|
172
|
+
delta=candidate[task_hash] - baseline[task_hash],
|
|
173
|
+
)
|
|
174
|
+
for task_hash in committed
|
|
175
|
+
]
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _rewards_by_task(
|
|
179
|
+
receipts: Sequence[EpisodeReceipt],
|
|
180
|
+
reward_name: str,
|
|
181
|
+
committed: Sequence[Digest],
|
|
182
|
+
label: str,
|
|
183
|
+
) -> dict[Digest, float]:
|
|
184
|
+
"""Read one variant's primary reward per task, refusing anything ambiguous."""
|
|
185
|
+
rewards: dict[Digest, float] = {}
|
|
186
|
+
for receipt in receipts:
|
|
187
|
+
traces = receipt.named_traces.get(SUBJECT_AGENT, [])
|
|
188
|
+
if len(traces) != 1:
|
|
189
|
+
raise VerificationError(
|
|
190
|
+
f"a {label} receipt carries {len(traces)} subject traces; one "
|
|
191
|
+
"episode has exactly one",
|
|
192
|
+
code=TASK_MEMBERSHIP_MISMATCH,
|
|
193
|
+
details={"task_hash": receipt.task_hash, "traces": len(traces)},
|
|
194
|
+
)
|
|
195
|
+
if receipt.task_hash in rewards:
|
|
196
|
+
raise VerificationError(
|
|
197
|
+
f"the {label} variant scored task {receipt.task_hash} twice, so "
|
|
198
|
+
"one of the two rewards would have to be discarded",
|
|
199
|
+
code=TASK_MEMBERSHIP_MISMATCH,
|
|
200
|
+
details={"variant": label, "task_hash": receipt.task_hash},
|
|
201
|
+
)
|
|
202
|
+
reward = traces[0].rewards.get(reward_name)
|
|
203
|
+
if reward is None:
|
|
204
|
+
raise VerificationError(
|
|
205
|
+
f"a {label} receipt records no {reward_name!r} reward, which is "
|
|
206
|
+
"the reward this comparison is decided on",
|
|
207
|
+
code=REWARD_MISSING,
|
|
208
|
+
details={"task_hash": receipt.task_hash, "reward": reward_name},
|
|
209
|
+
)
|
|
210
|
+
_require_finite(reward, label, receipt.task_hash)
|
|
211
|
+
rewards[receipt.task_hash] = reward
|
|
212
|
+
|
|
213
|
+
missing: list[JsonValue] = [value for value in committed if value not in rewards]
|
|
214
|
+
unexpected: list[JsonValue] = [
|
|
215
|
+
value for value in sorted(set(rewards) - set(committed))
|
|
216
|
+
]
|
|
217
|
+
if missing or unexpected:
|
|
218
|
+
raise VerificationError(
|
|
219
|
+
f"the {label} variant scored a different set of tasks than the "
|
|
220
|
+
"Campaign commits to",
|
|
221
|
+
code=TASK_MEMBERSHIP_MISMATCH,
|
|
222
|
+
details={"variant": label, "missing": missing, "unexpected": unexpected},
|
|
223
|
+
)
|
|
224
|
+
return rewards
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _require_finite(value: float, label: str, task_hash: Digest) -> None:
|
|
228
|
+
"""Refuse a reward that cannot be averaged or canonically written down."""
|
|
229
|
+
if math.isfinite(value):
|
|
230
|
+
return
|
|
231
|
+
raise VerificationError(
|
|
232
|
+
f"the {label} reward recorded for task {task_hash} is not finite, so it "
|
|
233
|
+
"is not a measurement",
|
|
234
|
+
code=REWARD_NON_FINITE,
|
|
235
|
+
details={"variant": label, "task_hash": task_hash, "value": repr(value)},
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
# ---------------------------------------------------------------------------
|
|
240
|
+
# Aggregation
|
|
241
|
+
# ---------------------------------------------------------------------------
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def aggregate_primary_result(
|
|
245
|
+
deltas: Sequence[TaskDelta], reward_name: str
|
|
246
|
+
) -> PrimaryUpliftResult:
|
|
247
|
+
"""Compute the headline result from the paired rows and nothing else."""
|
|
248
|
+
if not deltas:
|
|
249
|
+
raise VerificationError(
|
|
250
|
+
"an uplift result summarizes at least one paired task",
|
|
251
|
+
code=TASK_MEMBERSHIP_MISMATCH,
|
|
252
|
+
details={"reward": reward_name, "task_count": 0},
|
|
253
|
+
)
|
|
254
|
+
for delta in deltas:
|
|
255
|
+
_require_finite(delta.baseline_reward, "baseline", delta.task_hash)
|
|
256
|
+
_require_finite(delta.candidate_reward, "candidate", delta.task_hash)
|
|
257
|
+
|
|
258
|
+
baseline_mean = _mean(delta.baseline_reward for delta in deltas)
|
|
259
|
+
candidate_mean = _mean(delta.candidate_reward for delta in deltas)
|
|
260
|
+
absolute = candidate_mean - baseline_mean
|
|
261
|
+
for value, label in (
|
|
262
|
+
(baseline_mean, "baseline mean"),
|
|
263
|
+
(candidate_mean, "candidate mean"),
|
|
264
|
+
(absolute, "absolute delta"),
|
|
265
|
+
):
|
|
266
|
+
if not math.isfinite(value):
|
|
267
|
+
raise VerificationError(
|
|
268
|
+
f"the {label} over these rewards is not a finite number",
|
|
269
|
+
code=REWARD_NON_FINITE,
|
|
270
|
+
details={"reward": reward_name, "task_count": len(deltas)},
|
|
271
|
+
)
|
|
272
|
+
|
|
273
|
+
return PrimaryUpliftResult(
|
|
274
|
+
reward_name=reward_name,
|
|
275
|
+
baseline_mean=baseline_mean,
|
|
276
|
+
candidate_mean=candidate_mean,
|
|
277
|
+
absolute_delta=absolute,
|
|
278
|
+
# Section 7.10: null over a zero baseline. Any number here would be an
|
|
279
|
+
# invented one.
|
|
280
|
+
relative_delta=None if baseline_mean == 0.0 else absolute / baseline_mean,
|
|
281
|
+
wins=sum(
|
|
282
|
+
1 for delta in deltas if delta.candidate_reward > delta.baseline_reward
|
|
283
|
+
),
|
|
284
|
+
losses=sum(
|
|
285
|
+
1 for delta in deltas if delta.candidate_reward < delta.baseline_reward
|
|
286
|
+
),
|
|
287
|
+
ties=sum(
|
|
288
|
+
1 for delta in deltas if delta.candidate_reward == delta.baseline_reward
|
|
289
|
+
),
|
|
290
|
+
)
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
def _mean(values: Iterable[float]) -> float:
|
|
294
|
+
collected = list(values)
|
|
295
|
+
return sum(collected) / len(collected)
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
# ---------------------------------------------------------------------------
|
|
299
|
+
# The verdict
|
|
300
|
+
# ---------------------------------------------------------------------------
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def decide_uplift(
|
|
304
|
+
*,
|
|
305
|
+
campaign: CampaignSpec,
|
|
306
|
+
comparison: RealComparisonResult,
|
|
307
|
+
primary: PrimaryUpliftResult,
|
|
308
|
+
) -> UpliftDecision:
|
|
309
|
+
"""Apply the Campaign's own acceptance rules, and no others.
|
|
310
|
+
|
|
311
|
+
``inconclusive`` is reached when the Campaign predeclared no rule that can
|
|
312
|
+
decide: a scoring contract that neither requires the candidate to out-score
|
|
313
|
+
the baseline nor sets a minimum delta would accept a regression, so calling
|
|
314
|
+
such a result "accepted" would report a verdict nobody specified.
|
|
315
|
+
"""
|
|
316
|
+
if not comparison.controlled:
|
|
317
|
+
return UpliftDecision.INVALID
|
|
318
|
+
|
|
319
|
+
scoring = campaign.scoring
|
|
320
|
+
if not scoring.require_candidate_above_baseline and (
|
|
321
|
+
scoring.minimum_absolute_delta == 0.0
|
|
322
|
+
):
|
|
323
|
+
return UpliftDecision.INCONCLUSIVE
|
|
324
|
+
if scoring.require_candidate_above_baseline and primary.absolute_delta <= 0.0:
|
|
325
|
+
return UpliftDecision.REJECTED
|
|
326
|
+
if primary.absolute_delta < scoring.minimum_absolute_delta:
|
|
327
|
+
return UpliftDecision.REJECTED
|
|
328
|
+
return UpliftDecision.ACCEPTED
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def summarize_receipts(
|
|
332
|
+
baseline_receipts: Sequence[EpisodeReceipt],
|
|
333
|
+
candidate_receipts: Sequence[EpisodeReceipt],
|
|
334
|
+
) -> tuple[ScoreStatus, EvidenceStatus]:
|
|
335
|
+
"""Return the score and evidence statuses the whole comparison carries.
|
|
336
|
+
|
|
337
|
+
A comparison is only as good as its weakest receipt. One rollout whose
|
|
338
|
+
scoring errored makes the aggregate score invalid rather than making the
|
|
339
|
+
other rollouts' scores worth less, and one receipt whose evidence is
|
|
340
|
+
partial makes the comparison's evidence partial.
|
|
341
|
+
"""
|
|
342
|
+
receipts = [*baseline_receipts, *candidate_receipts]
|
|
343
|
+
if not receipts:
|
|
344
|
+
return ScoreStatus.MISSING, EvidenceStatus.NOT_COLLECTED
|
|
345
|
+
|
|
346
|
+
score = (
|
|
347
|
+
ScoreStatus.VALID
|
|
348
|
+
if all(receipt.score_status is ScoreStatus.VALID for receipt in receipts)
|
|
349
|
+
else ScoreStatus.INVALID
|
|
350
|
+
)
|
|
351
|
+
evidence = (
|
|
352
|
+
EvidenceStatus.COMPLETE
|
|
353
|
+
if all(
|
|
354
|
+
receipt.evidence_status is EvidenceStatus.COMPLETE for receipt in receipts
|
|
355
|
+
)
|
|
356
|
+
else EvidenceStatus.PARTIAL
|
|
357
|
+
)
|
|
358
|
+
return score, evidence
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
def publication_status_for(data_policy: DataPolicy) -> PublicationStatus:
|
|
362
|
+
"""Return where a fresh report stands with respect to being published.
|
|
363
|
+
|
|
364
|
+
Two of the five statuses can be true of a report that has just been
|
|
365
|
+
written, and which one it is comes from the rights statement the run
|
|
366
|
+
executed under rather than from anything the run did. A DataPolicy that
|
|
367
|
+
does not make the uplift report public is a policy under which publishing
|
|
368
|
+
it is not a thing anybody may choose: that is ``blocked``, decided once,
|
|
369
|
+
before anybody is offered anything. A policy that does make it public
|
|
370
|
+
leaves the choice open, and a choice nobody has made yet is
|
|
371
|
+
``not_requested``.
|
|
372
|
+
|
|
373
|
+
The other three describe an attempt rather than a report. A run's
|
|
374
|
+
publication journal owns those, because it is the only record of an
|
|
375
|
+
attempt; nothing rewrites a signed report to say it was published, and the
|
|
376
|
+
proof it belongs to is what makes that impossible as well as wrong.
|
|
377
|
+
"""
|
|
378
|
+
if data_policy.derived_artifacts.uplift_report == "public":
|
|
379
|
+
return PublicationStatus.NOT_REQUESTED
|
|
380
|
+
return PublicationStatus.BLOCKED
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def publication_eligible_for(
|
|
384
|
+
*,
|
|
385
|
+
grade: Literal["development_only", "P1"],
|
|
386
|
+
publication: PublicationStatus,
|
|
387
|
+
) -> bool:
|
|
388
|
+
"""Return whether this report may be published at all.
|
|
389
|
+
|
|
390
|
+
Until there was somewhere to publish to, this was a constant ``False``
|
|
391
|
+
whose comment said why: no route, no credential, no server. There is a
|
|
392
|
+
route now, so the flag has to answer the question it is named for, and
|
|
393
|
+
the answer has two halves and no third.
|
|
394
|
+
|
|
395
|
+
*The evidence has to be worth publishing.* A ``development_only`` report is
|
|
396
|
+
either invented numbers or a real comparison nothing sealed, and neither is
|
|
397
|
+
evidence of anything. Only a P1 report — signed receipts, a signed report,
|
|
398
|
+
the public key travelling with them, a controlled comparison and a valid
|
|
399
|
+
score — is eligible, which is the same bar decisions 0005 section 3.4 sets
|
|
400
|
+
for the grade itself.
|
|
401
|
+
|
|
402
|
+
*The rights have to permit it.* A report whose policy blocks publication is
|
|
403
|
+
not eligible however good the evidence is.
|
|
404
|
+
|
|
405
|
+
Both halves are also what the frozen model's own validator insists on, so
|
|
406
|
+
the computation cannot produce a report the model would refuse.
|
|
407
|
+
"""
|
|
408
|
+
return grade == "P1" and publication is not PublicationStatus.BLOCKED
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
def proof_grade_for(
|
|
412
|
+
*,
|
|
413
|
+
attestation: LocalAttestation,
|
|
414
|
+
comparison: ComparisonStatus,
|
|
415
|
+
score: ScoreStatus,
|
|
416
|
+
) -> Literal["development_only", "P1"]:
|
|
417
|
+
"""Return the strongest grade this report is entitled to claim.
|
|
418
|
+
|
|
419
|
+
Decisions document 0005 section 3.4. The signature conditions are the
|
|
420
|
+
caller's to establish and are summarized by ``attestation``; the two
|
|
421
|
+
conditions this function can check itself are checked here, so a signed
|
|
422
|
+
report over an uncontrolled comparison still cannot claim P1.
|
|
423
|
+
"""
|
|
424
|
+
controlled = comparison in (
|
|
425
|
+
ComparisonStatus.CONTROLLED,
|
|
426
|
+
ComparisonStatus.CONTROLLED_WITH_WARNINGS,
|
|
427
|
+
)
|
|
428
|
+
if (
|
|
429
|
+
attestation is LocalAttestation.LOCAL_ED25519
|
|
430
|
+
and controlled
|
|
431
|
+
and score is ScoreStatus.VALID
|
|
432
|
+
):
|
|
433
|
+
return "P1"
|
|
434
|
+
return "development_only"
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
# ---------------------------------------------------------------------------
|
|
438
|
+
# The report
|
|
439
|
+
# ---------------------------------------------------------------------------
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
def build_uplift_report(
|
|
443
|
+
*,
|
|
444
|
+
run_request: RunRequest,
|
|
445
|
+
campaign: CampaignSpec,
|
|
446
|
+
data_policy: DataPolicy,
|
|
447
|
+
taskset_validation_receipt_digest: Digest,
|
|
448
|
+
baseline_manifest: ExperimentManifest,
|
|
449
|
+
candidate_manifest: ExperimentManifest,
|
|
450
|
+
baseline_receipt_set: ReceiptSetManifest,
|
|
451
|
+
candidate_receipt_set: ReceiptSetManifest,
|
|
452
|
+
comparison: RealComparisonResult,
|
|
453
|
+
task_deltas: Sequence[TaskDelta],
|
|
454
|
+
primary: PrimaryUpliftResult,
|
|
455
|
+
score: ScoreStatus,
|
|
456
|
+
evidence: EvidenceStatus,
|
|
457
|
+
attestation: LocalAttestation,
|
|
458
|
+
created_at: datetime,
|
|
459
|
+
) -> UpliftReport:
|
|
460
|
+
"""Construct the canonical real local report, or refuse to construct one.
|
|
461
|
+
|
|
462
|
+
Two conditions are refusals rather than statuses. A comparison that is not
|
|
463
|
+
controlled did not measure the Skill, and a score that is not valid did not
|
|
464
|
+
measure anything; in both cases spec section 7.10's decision is ``invalid``,
|
|
465
|
+
and the frozen model has no way to carry that verdict without also claiming
|
|
466
|
+
the P1 grade that decisions document 0005 forbids an uncontrolled
|
|
467
|
+
comparison. A report saying "invalid" is worth less than a run that failed
|
|
468
|
+
with the reason, so the reason is raised.
|
|
469
|
+
|
|
470
|
+
``score`` and ``evidence`` are passed in rather than derived from the
|
|
471
|
+
receipt sets, which commit to receipts by digest and hold no statuses;
|
|
472
|
+
:func:`summarize_receipts` computes them from the receipts themselves.
|
|
473
|
+
|
|
474
|
+
The DataPolicy is passed in whole rather than by digest because publication
|
|
475
|
+
eligibility is read off its terms. It is checked against the digest the
|
|
476
|
+
run's request names, so a report cannot cite a Campaign's rights statement
|
|
477
|
+
and be graded under a different one.
|
|
478
|
+
"""
|
|
479
|
+
_require_reportable(comparison, score, run_request)
|
|
480
|
+
_require_lineage(
|
|
481
|
+
run_request=run_request,
|
|
482
|
+
campaign=campaign,
|
|
483
|
+
data_policy=data_policy,
|
|
484
|
+
baseline_manifest=baseline_manifest,
|
|
485
|
+
candidate_manifest=candidate_manifest,
|
|
486
|
+
baseline_receipt_set=baseline_receipt_set,
|
|
487
|
+
candidate_receipt_set=candidate_receipt_set,
|
|
488
|
+
comparison=comparison,
|
|
489
|
+
)
|
|
490
|
+
|
|
491
|
+
grade = proof_grade_for(
|
|
492
|
+
attestation=attestation, comparison=comparison.status, score=score
|
|
493
|
+
)
|
|
494
|
+
publication = publication_status_for(data_policy)
|
|
495
|
+
decision = (
|
|
496
|
+
decide_uplift(campaign=campaign, comparison=comparison, primary=primary)
|
|
497
|
+
if grade == "P1"
|
|
498
|
+
# An unsigned real report withholds the verdict rather than presenting
|
|
499
|
+
# one the frozen model would have to grade P1. See the module docstring.
|
|
500
|
+
else UpliftDecision.DEVELOPMENT_ONLY
|
|
501
|
+
)
|
|
502
|
+
|
|
503
|
+
return UpliftReport(
|
|
504
|
+
schema_version=UPLIFT_SCHEMA_VERSION,
|
|
505
|
+
id=new_id("uplift"),
|
|
506
|
+
run_id=run_request.run_id,
|
|
507
|
+
campaign_spec_digest=run_request.campaign_spec_digest,
|
|
508
|
+
program_ref=run_request.program_ref,
|
|
509
|
+
public_context=run_request.public_context,
|
|
510
|
+
data_policy_digest=run_request.data_policy_digest,
|
|
511
|
+
outcome_contract_digest=run_request.outcome_contract_digest,
|
|
512
|
+
evaluation_backend=campaign.evaluation_backend,
|
|
513
|
+
taskset_validation_receipt_digest=taskset_validation_receipt_digest,
|
|
514
|
+
baseline_manifest_digest=run_request.baseline_manifest_digest,
|
|
515
|
+
candidate_manifest_digest=run_request.candidate_manifest_digest,
|
|
516
|
+
statuses=UpliftStatuses(
|
|
517
|
+
# A report is only built for an execution that finished both
|
|
518
|
+
# variants; a partial one fails the run instead (spec 6.17).
|
|
519
|
+
execution=ExecutionStatus.COMPLETED,
|
|
520
|
+
score=score,
|
|
521
|
+
evidence=evidence,
|
|
522
|
+
comparison=comparison.status,
|
|
523
|
+
# Nothing has been uploaded: a report is written before anybody has
|
|
524
|
+
# been asked whether to publish it, so the only two answers
|
|
525
|
+
# available here are "nobody has asked" and "the rights forbid it".
|
|
526
|
+
# Spec section 7.10.
|
|
527
|
+
publication=publication,
|
|
528
|
+
),
|
|
529
|
+
manifest_comparison=comparison.manifest_comparison,
|
|
530
|
+
primary_result=primary,
|
|
531
|
+
task_deltas=list(task_deltas),
|
|
532
|
+
decision=decision,
|
|
533
|
+
proof_grade=grade,
|
|
534
|
+
publication_eligible=publication_eligible_for(
|
|
535
|
+
grade=grade, publication=publication
|
|
536
|
+
),
|
|
537
|
+
created_at=created_at,
|
|
538
|
+
)
|
|
539
|
+
|
|
540
|
+
|
|
541
|
+
def _require_reportable(
|
|
542
|
+
comparison: RealComparisonResult, score: ScoreStatus, run_request: RunRequest
|
|
543
|
+
) -> None:
|
|
544
|
+
"""Refuse to write a report over evidence that decided nothing."""
|
|
545
|
+
if not comparison.controlled:
|
|
546
|
+
raise VerificationError(
|
|
547
|
+
"this run's two variants were not one controlled experiment, so "
|
|
548
|
+
"there is no uplift to report: "
|
|
549
|
+
+ "; ".join(check.detail for check in comparison.failures),
|
|
550
|
+
code=COMPARISON_INVALID,
|
|
551
|
+
details={
|
|
552
|
+
"run_id": run_request.run_id,
|
|
553
|
+
"comparison": comparison.status.value,
|
|
554
|
+
"failed_checks": [check.id for check in comparison.failures],
|
|
555
|
+
},
|
|
556
|
+
)
|
|
557
|
+
if score is not ScoreStatus.VALID:
|
|
558
|
+
raise VerificationError(
|
|
559
|
+
f"this run's recorded scores are {score.value}, so the comparison "
|
|
560
|
+
"measured nothing that may be reported",
|
|
561
|
+
code=COMPARISON_INVALID,
|
|
562
|
+
details={"run_id": run_request.run_id, "score": score.value},
|
|
563
|
+
)
|
|
564
|
+
|
|
565
|
+
|
|
566
|
+
def _require_lineage(
|
|
567
|
+
*,
|
|
568
|
+
run_request: RunRequest,
|
|
569
|
+
campaign: CampaignSpec,
|
|
570
|
+
data_policy: DataPolicy,
|
|
571
|
+
baseline_manifest: ExperimentManifest,
|
|
572
|
+
candidate_manifest: ExperimentManifest,
|
|
573
|
+
baseline_receipt_set: ReceiptSetManifest,
|
|
574
|
+
candidate_receipt_set: ReceiptSetManifest,
|
|
575
|
+
comparison: RealComparisonResult,
|
|
576
|
+
) -> None:
|
|
577
|
+
"""Require every object the report cites to belong to the same run.
|
|
578
|
+
|
|
579
|
+
The receipt sets are inputs to this check rather than fields of the report:
|
|
580
|
+
the frozen report has nowhere to carry a receipt-set digest, and what it can
|
|
581
|
+
still do is refuse to summarize receipts that belong to a different run,
|
|
582
|
+
variant or manifest.
|
|
583
|
+
"""
|
|
584
|
+
campaign_digest = digest_object(campaign)
|
|
585
|
+
for label, expected, found in (
|
|
586
|
+
("Campaign", run_request.campaign_spec_digest, campaign_digest),
|
|
587
|
+
(
|
|
588
|
+
"baseline manifest",
|
|
589
|
+
run_request.baseline_manifest_digest,
|
|
590
|
+
digest_object(baseline_manifest),
|
|
591
|
+
),
|
|
592
|
+
(
|
|
593
|
+
"candidate manifest",
|
|
594
|
+
run_request.candidate_manifest_digest,
|
|
595
|
+
digest_object(candidate_manifest),
|
|
596
|
+
),
|
|
597
|
+
(
|
|
598
|
+
"DataPolicy",
|
|
599
|
+
run_request.data_policy_digest,
|
|
600
|
+
campaign.data_policy_digest,
|
|
601
|
+
),
|
|
602
|
+
# The document itself, and not only the digest the Campaign names.
|
|
603
|
+
# Publication eligibility is read off these terms, so the terms have to
|
|
604
|
+
# be the ones this run was executed under.
|
|
605
|
+
(
|
|
606
|
+
"DataPolicy document",
|
|
607
|
+
run_request.data_policy_digest,
|
|
608
|
+
digest_object(data_policy),
|
|
609
|
+
),
|
|
610
|
+
):
|
|
611
|
+
if expected != found:
|
|
612
|
+
raise VerificationError(
|
|
613
|
+
f"the {label} this report would cite is not the one the run's "
|
|
614
|
+
"request names",
|
|
615
|
+
code=COMPARISON_INVALID,
|
|
616
|
+
details={
|
|
617
|
+
"run_id": run_request.run_id,
|
|
618
|
+
"expected": expected,
|
|
619
|
+
"found": found,
|
|
620
|
+
},
|
|
621
|
+
)
|
|
622
|
+
|
|
623
|
+
committed = list(comparison.ordered_task_hashes)
|
|
624
|
+
for manifest_set, variant, manifest_digest in (
|
|
625
|
+
(
|
|
626
|
+
baseline_receipt_set,
|
|
627
|
+
ExperimentVariant.BASELINE,
|
|
628
|
+
run_request.baseline_manifest_digest,
|
|
629
|
+
),
|
|
630
|
+
(
|
|
631
|
+
candidate_receipt_set,
|
|
632
|
+
ExperimentVariant.CANDIDATE,
|
|
633
|
+
run_request.candidate_manifest_digest,
|
|
634
|
+
),
|
|
635
|
+
):
|
|
636
|
+
if (
|
|
637
|
+
manifest_set.run_id == run_request.run_id
|
|
638
|
+
and manifest_set.variant is variant
|
|
639
|
+
and manifest_set.experiment_manifest_digest == manifest_digest
|
|
640
|
+
and manifest_set.receipt_count == len(committed)
|
|
641
|
+
and manifest_set.task_membership_digest == membership_digest(committed)
|
|
642
|
+
):
|
|
643
|
+
continue
|
|
644
|
+
raise VerificationError(
|
|
645
|
+
f"the {variant.value} receipt set does not commit to this run's "
|
|
646
|
+
f"{variant.value} receipts",
|
|
647
|
+
code=COMPARISON_INVALID,
|
|
648
|
+
details={
|
|
649
|
+
"run_id": run_request.run_id,
|
|
650
|
+
"variant": variant.value,
|
|
651
|
+
"receipt_set_run_id": manifest_set.run_id,
|
|
652
|
+
"receipt_count": manifest_set.receipt_count,
|
|
653
|
+
"committed": len(committed),
|
|
654
|
+
},
|
|
655
|
+
)
|