techtree 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- techtree/__init__.py +35 -0
- techtree/__main__.py +14 -0
- techtree/canonical.py +239 -0
- techtree/catalog/__init__.py +25 -0
- techtree/catalog/repository.py +400 -0
- techtree/catalog/service.py +419 -0
- techtree/cli/__init__.py +1 -0
- techtree/cli/app.py +416 -0
- techtree/cli/commands/__init__.py +1 -0
- techtree/cli/commands/climb.py +1223 -0
- techtree/cli/commands/doctor.py +147 -0
- techtree/cli/commands/engine.py +207 -0
- techtree/cli/commands/proof.py +556 -0
- techtree/cli/commands/publish.py +447 -0
- techtree/cli/commands/release.py +303 -0
- techtree/cli/commands/run.py +1067 -0
- techtree/cli/commands/setup.py +181 -0
- techtree/cli/commands/skill.py +221 -0
- techtree/cli/commands/uplift.py +698 -0
- techtree/cli/commands/withdraw.py +212 -0
- techtree/cli/confirm.py +47 -0
- techtree/cli/context.py +96 -0
- techtree/cli/invoke.py +220 -0
- techtree/cli/output.py +280 -0
- techtree/constants.py +138 -0
- techtree/crypto.py +128 -0
- techtree/doctor/__init__.py +1 -0
- techtree/doctor/checks.py +675 -0
- techtree/doctor/execution_checks.py +435 -0
- techtree/doctor/service.py +326 -0
- techtree/drafts/__init__.py +32 -0
- techtree/drafts/source.py +146 -0
- techtree/drafts/store.py +992 -0
- techtree/engines/__init__.py +1 -0
- techtree/engines/bundle.py +251 -0
- techtree/engines/installer.py +679 -0
- techtree/engines/registry.py +235 -0
- techtree/engines/runner.py +170 -0
- techtree/errors.py +262 -0
- techtree/fs.py +234 -0
- techtree/harness.py +108 -0
- techtree/identity/__init__.py +41 -0
- techtree/identity/models.py +113 -0
- techtree/identity/service.py +199 -0
- techtree/identity/store.py +263 -0
- techtree/ids.py +85 -0
- techtree/manifests/__init__.py +39 -0
- techtree/manifests/builder.py +433 -0
- techtree/manifests/compare.py +376 -0
- techtree/models/__init__.py +282 -0
- techtree/models/base.py +201 -0
- techtree/models/campaign.py +484 -0
- techtree/models/catalog.py +227 -0
- techtree/models/cli.py +151 -0
- techtree/models/climb.py +254 -0
- techtree/models/data_policy.py +130 -0
- techtree/models/engine.py +156 -0
- techtree/models/episode_receipt.py +130 -0
- techtree/models/evaluation_backend.py +113 -0
- techtree/models/experiment.py +154 -0
- techtree/models/run.py +214 -0
- techtree/models/skill.py +156 -0
- techtree/models/uplift_report.py +158 -0
- techtree/models/validation.py +299 -0
- techtree/paths.py +116 -0
- techtree/presentation/__init__.py +31 -0
- techtree/presentation/build.py +1242 -0
- techtree/presentation/compact.py +246 -0
- techtree/presentation/evidence.py +169 -0
- techtree/presentation/models.py +358 -0
- techtree/presentation/rich.py +312 -0
- techtree/presentation/sanitize.py +156 -0
- techtree/publication/__init__.py +44 -0
- techtree/publication/address.py +180 -0
- techtree/publication/coordinates.py +26 -0
- techtree/publication/journal.py +212 -0
- techtree/publication/keccak.py +183 -0
- techtree/publication/models.py +209 -0
- techtree/publication/offer.py +35 -0
- techtree/publication/service.py +618 -0
- techtree/publication/transport.py +296 -0
- techtree/publication/verify.py +242 -0
- techtree/publication/withdraw.py +156 -0
- techtree/py.typed +0 -0
- techtree/receipts/__init__.py +52 -0
- techtree/receipts/bundle.py +578 -0
- techtree/receipts/compare.py +1065 -0
- techtree/receipts/episode.py +672 -0
- techtree/receipts/execution.py +630 -0
- techtree/receipts/observed.py +474 -0
- techtree/receipts/set.py +336 -0
- techtree/receipts/uplift.py +655 -0
- techtree/receipts/verify.py +1055 -0
- techtree/release/__init__.py +9 -0
- techtree/release/bootstrap.py +509 -0
- techtree/release/checks.py +376 -0
- techtree/release/document.py +125 -0
- techtree/release/generate.py +221 -0
- techtree/release/models.py +293 -0
- techtree/release/provenance.py +109 -0
- techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
- techtree/resources/catalog/catalog.json +32 -0
- techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
- techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
- techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
- techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
- techtree/resources/engines/default/engine.json +20 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
- techtree/resources/engines/default/pyproject.toml +23 -0
- techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
- techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
- techtree/resources/engines/default/tools/normalize_validation.py +222 -0
- techtree/resources/engines/default/uv.lock +1758 -0
- techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
- techtree/resources/release/build-provenance.json +4 -0
- techtree/resources/release/release-core.json +24 -0
- techtree/runs/__init__.py +31 -0
- techtree/runs/artifacts.py +750 -0
- techtree/runs/child_registry.py +228 -0
- techtree/runs/events.py +478 -0
- techtree/runs/executor.py +140 -0
- techtree/runs/fake.py +741 -0
- techtree/runs/launcher.py +253 -0
- techtree/runs/machine.py +489 -0
- techtree/runs/real.py +789 -0
- techtree/runs/service.py +616 -0
- techtree/runs/store.py +555 -0
- techtree/runs/validation.py +259 -0
- techtree/runs/variants.py +684 -0
- techtree/settings.py +143 -0
- techtree/skills/__init__.py +14 -0
- techtree/skills/archive.py +282 -0
- techtree/skills/policy.py +62 -0
- techtree/skills/scanner.py +394 -0
- techtree/skills/service.py +752 -0
- techtree/skills/starter.py +434 -0
- techtree/tasksets/__init__.py +1 -0
- techtree/tasksets/membership.py +269 -0
- techtree/tasksets/provider.py +207 -0
- techtree/tasksets/resolver.py +311 -0
- techtree/tasksets/service.py +484 -0
- techtree/tasksets/verifiers_cli.py +538 -0
- techtree/uplift/__init__.py +20 -0
- techtree/uplift/context.py +544 -0
- techtree/uplift/derive.py +203 -0
- techtree/uplift/public_tasks.py +151 -0
- techtree/uplift/service.py +719 -0
- techtree/uplift/source.py +160 -0
- techtree/verifiers/__init__.py +31 -0
- techtree/verifiers/budget.py +219 -0
- techtree/verifiers/child.py +633 -0
- techtree/verifiers/compiler.py +432 -0
- techtree/verifiers/config.py +365 -0
- techtree/verifiers/credentials.py +321 -0
- techtree/verifiers/image.py +126 -0
- techtree/verifiers/models.py +527 -0
- techtree/verifiers/outputs.py +368 -0
- techtree/verifiers/progress.py +192 -0
- techtree/verifiers/supervisor.py +341 -0
- techtree/verifiers/verify.py +782 -0
- techtree/version.py +39 -0
- techtree/worker/__init__.py +18 -0
- techtree/worker/execute.py +487 -0
- techtree/worker/main.py +57 -0
- techtree-0.1.0.dist-info/METADATA +344 -0
- techtree-0.1.0.dist-info/RECORD +174 -0
- techtree-0.1.0.dist-info/WHEEL +4 -0
- techtree-0.1.0.dist-info/entry_points.txt +3 -0
- techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,1055 @@
|
|
|
1
|
+
"""Checking a local proof, offline, from exactly the bytes it stored.
|
|
2
|
+
|
|
3
|
+
Spec section 7.12, and the verification order fixed by section 7.11.
|
|
4
|
+
|
|
5
|
+
Nothing in this module reaches the network, the catalog, the settings file, or
|
|
6
|
+
the machine's own identity. It is handed a directory and it answers one
|
|
7
|
+
question about it: do these files still say what they said, and do they still
|
|
8
|
+
agree with each other? That is the whole of what a self-issued key can
|
|
9
|
+
establish, and stating it precisely is what keeps ``P1`` from drifting into
|
|
10
|
+
sounding like something a third party checked.
|
|
11
|
+
|
|
12
|
+
The order is the specification's, and each step depends on the one before it:
|
|
13
|
+
|
|
14
|
+
```text
|
|
15
|
+
1. Validate the bundle manifest.
|
|
16
|
+
2. Verify every artifact digest.
|
|
17
|
+
3. Verify Campaign and policy linkage.
|
|
18
|
+
4. Verify TasksetLock and validation-receipt linkage.
|
|
19
|
+
5. Verify every EpisodeReceipt envelope signature.
|
|
20
|
+
6. Verify the receipt sets.
|
|
21
|
+
7. Verify the UpliftReport envelope signature.
|
|
22
|
+
8. Recompute the paired aggregate from the receipts.
|
|
23
|
+
9. Require the recomputed result to equal the report.
|
|
24
|
+
10. Require the report's publication fields to hold together.
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
Two habits make those steps mean something.
|
|
28
|
+
|
|
29
|
+
*Everything is recomputed.* A digest recorded beside the thing it describes is
|
|
30
|
+
worth nothing on its own; every digest here is taken again from the file's own
|
|
31
|
+
bytes, and every aggregate is recomputed from the receipts rather than read out
|
|
32
|
+
of the report it is supposed to check.
|
|
33
|
+
|
|
34
|
+
*Nothing raises past the first problem.* A reader whose bundle is broken wants
|
|
35
|
+
to know everything that is broken about it, so every step records a named check
|
|
36
|
+
and the verdict is computed from the collected checks. The typed section 15
|
|
37
|
+
errors are then raised by the caller — the CLI — which knows whether the reader
|
|
38
|
+
asked for a verdict or for a report.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
from __future__ import annotations
|
|
42
|
+
|
|
43
|
+
from collections.abc import Callable, Sequence
|
|
44
|
+
from pathlib import Path
|
|
45
|
+
from typing import Final
|
|
46
|
+
|
|
47
|
+
from pydantic import BaseModel
|
|
48
|
+
from pydantic import ValidationError as PydanticValidationError
|
|
49
|
+
|
|
50
|
+
from techtree.canonical import digest_object, sha256_digest_bytes
|
|
51
|
+
from techtree.errors import VerificationError
|
|
52
|
+
from techtree.identity.models import (
|
|
53
|
+
LOCAL_IDENTITY_INVALID,
|
|
54
|
+
SIGNATURE_VERIFICATION_FAILED,
|
|
55
|
+
ExecutorIdentity,
|
|
56
|
+
VerificationMessage,
|
|
57
|
+
VerificationResult,
|
|
58
|
+
VerificationStatus,
|
|
59
|
+
)
|
|
60
|
+
from techtree.identity.service import verify_signed_object
|
|
61
|
+
from techtree.models.base import ObjectEnvelope
|
|
62
|
+
from techtree.models.campaign import CampaignSpec
|
|
63
|
+
from techtree.models.data_policy import DataPolicy
|
|
64
|
+
from techtree.models.episode_receipt import EpisodeReceipt
|
|
65
|
+
from techtree.models.experiment import ExperimentManifest, ExperimentVariant
|
|
66
|
+
from techtree.models.uplift_report import (
|
|
67
|
+
ComparisonStatus,
|
|
68
|
+
PublicationStatus,
|
|
69
|
+
UpliftReport,
|
|
70
|
+
)
|
|
71
|
+
from techtree.models.validation import TasksetLock, TasksetValidationReceipt
|
|
72
|
+
from techtree.receipts.bundle import (
|
|
73
|
+
BUNDLE_MANIFEST_FILENAME,
|
|
74
|
+
CAMPAIGN_FILENAME,
|
|
75
|
+
DATA_POLICY_FILENAME,
|
|
76
|
+
P1_ARTIFACT_DIGESTS_VERIFY,
|
|
77
|
+
P1_COMPARISON_CONTROLLED,
|
|
78
|
+
P1_PUBLIC_KEY_PRESENT,
|
|
79
|
+
P1_RECEIPTS_SIGNED,
|
|
80
|
+
P1_REPORT_SIGNED,
|
|
81
|
+
P1_SCORE_VALID,
|
|
82
|
+
PROOF_BUNDLE_INVALID,
|
|
83
|
+
PUBLIC_IDENTITY_FILENAME,
|
|
84
|
+
REPORT_FILENAME,
|
|
85
|
+
TASKSET_LOCK_FILENAME,
|
|
86
|
+
VALIDATION_RECEIPT_FILENAME,
|
|
87
|
+
LocalProofBundleManifest,
|
|
88
|
+
experiment_filename,
|
|
89
|
+
receipt_filename,
|
|
90
|
+
receipt_set_filename,
|
|
91
|
+
)
|
|
92
|
+
from techtree.receipts.compare import COMPARISON_INVALID
|
|
93
|
+
from techtree.receipts.execution import (
|
|
94
|
+
COMPARISON_EXECUTION_RECORD_INVALID,
|
|
95
|
+
EXECUTION_RECORD_FILENAME,
|
|
96
|
+
OPERATIONAL_EVIDENCE_UNAVAILABLE,
|
|
97
|
+
ComparisonExecutionRecord,
|
|
98
|
+
)
|
|
99
|
+
from techtree.receipts.set import (
|
|
100
|
+
RECEIPT_SET_INVALID,
|
|
101
|
+
ReceiptSetManifest,
|
|
102
|
+
verify_receipt_set,
|
|
103
|
+
)
|
|
104
|
+
from techtree.receipts.uplift import (
|
|
105
|
+
aggregate_primary_result,
|
|
106
|
+
pair_task_rewards,
|
|
107
|
+
publication_eligible_for,
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
__all__ = [
|
|
111
|
+
"LocalProofVerifier",
|
|
112
|
+
"verify_local_bundle",
|
|
113
|
+
"verify_report_envelope",
|
|
114
|
+
]
|
|
115
|
+
|
|
116
|
+
_PASSED: Final = "passed"
|
|
117
|
+
_FAILED: Final = "failed"
|
|
118
|
+
_WARNING: Final = "warning"
|
|
119
|
+
|
|
120
|
+
_VARIANT_ORDER: Final[tuple[ExperimentVariant, ...]] = (
|
|
121
|
+
ExperimentVariant.BASELINE,
|
|
122
|
+
ExperimentVariant.CANDIDATE,
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
#: What a P1 report claims, in the only words decisions document 0005 permits.
|
|
126
|
+
P1_MEANING: Final = "integrity-bound, participant-attested local execution"
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
class LocalProofVerifier:
|
|
130
|
+
"""Verifies local proofs without needing anything but their own bytes."""
|
|
131
|
+
|
|
132
|
+
def verify_report(self, path: Path) -> VerificationResult:
|
|
133
|
+
"""Verify one signed UpliftReport envelope against the key beside it."""
|
|
134
|
+
return verify_report_envelope(path)
|
|
135
|
+
|
|
136
|
+
def verify_bundle(self, path: Path) -> VerificationResult:
|
|
137
|
+
"""Verify a whole proof bundle, in the section 7.11 order."""
|
|
138
|
+
return verify_local_bundle(path)
|
|
139
|
+
|
|
140
|
+
def explain(self, result: VerificationResult) -> list[VerificationMessage]:
|
|
141
|
+
"""Summarize a verification under the five headings a reader needs.
|
|
142
|
+
|
|
143
|
+
Spec section 7.12 requires human output to keep these apart, because
|
|
144
|
+
collapsing them is exactly how "the signature verifies" turns into
|
|
145
|
+
"the result is proven".
|
|
146
|
+
"""
|
|
147
|
+
return _explain(result)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
# ---------------------------------------------------------------------------
|
|
151
|
+
# One envelope
|
|
152
|
+
# ---------------------------------------------------------------------------
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def verify_report_envelope(path: Path) -> VerificationResult:
|
|
156
|
+
"""Verify a signed report envelope using the public key stored beside it.
|
|
157
|
+
|
|
158
|
+
A bare envelope carries a signature and a key identifier, and no key. The
|
|
159
|
+
public half lives next to it in the bundle layout, so that is where this
|
|
160
|
+
looks; a report handed over without it cannot be checked at all, and saying
|
|
161
|
+
so is more useful than reporting a signature as unverifiable.
|
|
162
|
+
"""
|
|
163
|
+
checks = _Checks()
|
|
164
|
+
envelope = _load_envelope(path, UpliftReport, checks, "uplift-report")
|
|
165
|
+
identity = _load_identity(path.parent / PUBLIC_IDENTITY_FILENAME, checks)
|
|
166
|
+
if envelope is None or identity is None:
|
|
167
|
+
return checks.result()
|
|
168
|
+
|
|
169
|
+
checks.extend(
|
|
170
|
+
verify_signed_object(
|
|
171
|
+
identity=identity, envelope=envelope, subject="uplift-report"
|
|
172
|
+
).messages
|
|
173
|
+
)
|
|
174
|
+
_check_publication(envelope.payload, checks)
|
|
175
|
+
return checks.result()
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
# ---------------------------------------------------------------------------
|
|
179
|
+
# A whole bundle
|
|
180
|
+
# ---------------------------------------------------------------------------
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def verify_local_bundle(path: Path) -> VerificationResult:
|
|
184
|
+
"""Verify one proof bundle offline and return every check it ran."""
|
|
185
|
+
checks = _Checks()
|
|
186
|
+
directory = path if path.is_dir() else path.parent
|
|
187
|
+
|
|
188
|
+
# 1. The manifest, and the key it says signed everything.
|
|
189
|
+
sealed = _load_envelope(
|
|
190
|
+
directory / BUNDLE_MANIFEST_FILENAME,
|
|
191
|
+
LocalProofBundleManifest,
|
|
192
|
+
checks,
|
|
193
|
+
"bundle",
|
|
194
|
+
)
|
|
195
|
+
if sealed is None:
|
|
196
|
+
return checks.result()
|
|
197
|
+
manifest = sealed.payload
|
|
198
|
+
identity = manifest.executor_identity
|
|
199
|
+
checks.extend(
|
|
200
|
+
verify_signed_object(
|
|
201
|
+
identity=identity, envelope=sealed, subject="bundle"
|
|
202
|
+
).messages
|
|
203
|
+
)
|
|
204
|
+
|
|
205
|
+
# 2. Every artifact digest, recomputed from the stored file.
|
|
206
|
+
_check_artifacts(directory, manifest, checks)
|
|
207
|
+
|
|
208
|
+
# The public key travels as its own file, and it must be the same key.
|
|
209
|
+
stored_identity = _load_identity(directory / PUBLIC_IDENTITY_FILENAME, checks)
|
|
210
|
+
checks.record(
|
|
211
|
+
"bundle.public_key",
|
|
212
|
+
_PASSED if stored_identity == identity else _FAILED,
|
|
213
|
+
LOCAL_IDENTITY_INVALID,
|
|
214
|
+
(
|
|
215
|
+
f"the bundle carries public key {identity.key_id}"
|
|
216
|
+
if stored_identity == identity
|
|
217
|
+
else "the stored public key is not the key the manifest names"
|
|
218
|
+
),
|
|
219
|
+
)
|
|
220
|
+
|
|
221
|
+
documents = _load_documents(directory, checks)
|
|
222
|
+
if documents is None:
|
|
223
|
+
return checks.result()
|
|
224
|
+
|
|
225
|
+
# 3-4. Lineage: Campaign, policy, lock, validation receipt.
|
|
226
|
+
_check_linkage(manifest, documents, checks)
|
|
227
|
+
|
|
228
|
+
# 5-6. Receipts and the commitments over them.
|
|
229
|
+
receipts = _check_receipts(directory, documents, identity, checks)
|
|
230
|
+
|
|
231
|
+
# 7. The report envelope itself.
|
|
232
|
+
checks.extend(
|
|
233
|
+
verify_signed_object(
|
|
234
|
+
identity=identity, envelope=documents.report, subject="uplift-report"
|
|
235
|
+
).messages
|
|
236
|
+
)
|
|
237
|
+
checks.record(
|
|
238
|
+
"bundle.root_report_digest",
|
|
239
|
+
_PASSED
|
|
240
|
+
if manifest.root_report_digest == documents.report.payload_digest
|
|
241
|
+
else _FAILED,
|
|
242
|
+
PROOF_BUNDLE_INVALID,
|
|
243
|
+
"the manifest's root report digest names the report it carries"
|
|
244
|
+
if manifest.root_report_digest == documents.report.payload_digest
|
|
245
|
+
else "the manifest names a different report than the one it carries",
|
|
246
|
+
)
|
|
247
|
+
|
|
248
|
+
# 8-9. The aggregate, recomputed from the receipts.
|
|
249
|
+
if receipts is not None:
|
|
250
|
+
_check_aggregate(documents, receipts, checks)
|
|
251
|
+
|
|
252
|
+
# 10. Publication, which never happened and could not have.
|
|
253
|
+
_check_publication(documents.report.payload, checks)
|
|
254
|
+
|
|
255
|
+
# The operational record, if this run produced one. Decisions 0007 R6: it
|
|
256
|
+
# is checked as carefully as everything else and its absence is a warning,
|
|
257
|
+
# because it says what the comparison consumed rather than what it proved.
|
|
258
|
+
_check_execution_record(directory, manifest, documents, identity, checks)
|
|
259
|
+
|
|
260
|
+
# The decisions-0005 section 3.4 conditions, re-derived from these bytes.
|
|
261
|
+
_check_p1_conditions(
|
|
262
|
+
manifest=manifest,
|
|
263
|
+
documents=documents,
|
|
264
|
+
receipts=receipts,
|
|
265
|
+
identity_matches=stored_identity == identity,
|
|
266
|
+
checks=checks,
|
|
267
|
+
)
|
|
268
|
+
return checks.result()
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
# ---------------------------------------------------------------------------
|
|
272
|
+
# The pieces
|
|
273
|
+
# ---------------------------------------------------------------------------
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
class _Documents:
|
|
277
|
+
"""Every parsed document one bundle carries."""
|
|
278
|
+
|
|
279
|
+
def __init__(
|
|
280
|
+
self,
|
|
281
|
+
*,
|
|
282
|
+
campaign: CampaignSpec,
|
|
283
|
+
data_policy: DataPolicy,
|
|
284
|
+
taskset_lock: TasksetLock,
|
|
285
|
+
validation_receipt: TasksetValidationReceipt,
|
|
286
|
+
experiments: dict[ExperimentVariant, ExperimentManifest],
|
|
287
|
+
receipt_sets: dict[ExperimentVariant, ReceiptSetManifest],
|
|
288
|
+
report: ObjectEnvelope[UpliftReport],
|
|
289
|
+
) -> None:
|
|
290
|
+
self.campaign = campaign
|
|
291
|
+
self.data_policy = data_policy
|
|
292
|
+
self.taskset_lock = taskset_lock
|
|
293
|
+
self.validation_receipt = validation_receipt
|
|
294
|
+
self.experiments = experiments
|
|
295
|
+
self.receipt_sets = receipt_sets
|
|
296
|
+
self.report = report
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def _load_documents(directory: Path, checks: _Checks) -> _Documents | None:
|
|
300
|
+
"""Parse every document the bundle layout requires, or say which is unreadable."""
|
|
301
|
+
campaign = _load_model(directory / CAMPAIGN_FILENAME, CampaignSpec, checks)
|
|
302
|
+
policy = _load_model(directory / DATA_POLICY_FILENAME, DataPolicy, checks)
|
|
303
|
+
lock = _load_model(directory / TASKSET_LOCK_FILENAME, TasksetLock, checks)
|
|
304
|
+
receipt = _load_model(
|
|
305
|
+
directory / VALIDATION_RECEIPT_FILENAME, TasksetValidationReceipt, checks
|
|
306
|
+
)
|
|
307
|
+
report = _load_envelope(
|
|
308
|
+
directory / REPORT_FILENAME, UpliftReport, checks, "uplift-report"
|
|
309
|
+
)
|
|
310
|
+
experiments = {
|
|
311
|
+
variant: _load_model(
|
|
312
|
+
directory / experiment_filename(variant), ExperimentManifest, checks
|
|
313
|
+
)
|
|
314
|
+
for variant in _VARIANT_ORDER
|
|
315
|
+
}
|
|
316
|
+
receipt_sets = {
|
|
317
|
+
variant: _load_model(
|
|
318
|
+
directory / receipt_set_filename(variant), ReceiptSetManifest, checks
|
|
319
|
+
)
|
|
320
|
+
for variant in _VARIANT_ORDER
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
if (
|
|
324
|
+
campaign is None
|
|
325
|
+
or policy is None
|
|
326
|
+
or lock is None
|
|
327
|
+
or receipt is None
|
|
328
|
+
or report is None
|
|
329
|
+
or any(value is None for value in experiments.values())
|
|
330
|
+
or any(value is None for value in receipt_sets.values())
|
|
331
|
+
):
|
|
332
|
+
return None
|
|
333
|
+
return _Documents(
|
|
334
|
+
campaign=campaign,
|
|
335
|
+
data_policy=policy,
|
|
336
|
+
taskset_lock=lock,
|
|
337
|
+
validation_receipt=receipt,
|
|
338
|
+
experiments={
|
|
339
|
+
variant: value
|
|
340
|
+
for variant, value in experiments.items()
|
|
341
|
+
if value is not None
|
|
342
|
+
},
|
|
343
|
+
receipt_sets={
|
|
344
|
+
variant: value
|
|
345
|
+
for variant, value in receipt_sets.items()
|
|
346
|
+
if value is not None
|
|
347
|
+
},
|
|
348
|
+
report=report,
|
|
349
|
+
)
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
def _check_artifacts(
|
|
353
|
+
directory: Path, manifest: LocalProofBundleManifest, checks: _Checks
|
|
354
|
+
) -> None:
|
|
355
|
+
"""Recompute every placed artifact's digest and size from its own bytes."""
|
|
356
|
+
for reference in manifest.artifacts:
|
|
357
|
+
relative_path = reference.relative_path
|
|
358
|
+
assert relative_path is not None # the manifest's own validator requires it
|
|
359
|
+
stored = directory / relative_path
|
|
360
|
+
try:
|
|
361
|
+
data = stored.read_bytes()
|
|
362
|
+
except OSError:
|
|
363
|
+
checks.record(
|
|
364
|
+
f"artifact.{relative_path}",
|
|
365
|
+
_FAILED,
|
|
366
|
+
PROOF_BUNDLE_INVALID,
|
|
367
|
+
f"the bundle names {relative_path}, which is not there",
|
|
368
|
+
)
|
|
369
|
+
continue
|
|
370
|
+
digest = sha256_digest_bytes(data)
|
|
371
|
+
matches = digest == reference.digest and len(data) == reference.size
|
|
372
|
+
checks.record(
|
|
373
|
+
f"artifact.{relative_path}",
|
|
374
|
+
_PASSED if matches else _FAILED,
|
|
375
|
+
PROOF_BUNDLE_INVALID,
|
|
376
|
+
f"{relative_path} matches the digest the bundle commits to"
|
|
377
|
+
if matches
|
|
378
|
+
else (
|
|
379
|
+
f"{relative_path} has changed since the bundle was written: "
|
|
380
|
+
f"committed {reference.digest}, stored {digest}"
|
|
381
|
+
),
|
|
382
|
+
)
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def _check_linkage(
|
|
386
|
+
manifest: LocalProofBundleManifest, documents: _Documents, checks: _Checks
|
|
387
|
+
) -> None:
|
|
388
|
+
"""Check every edge between the documents, in both directions."""
|
|
389
|
+
report = documents.report.payload
|
|
390
|
+
campaign_digest = digest_object(documents.campaign)
|
|
391
|
+
policy_digest = digest_object(documents.data_policy)
|
|
392
|
+
lock_digest = digest_object(documents.taskset_lock)
|
|
393
|
+
receipt_digest = digest_object(documents.validation_receipt)
|
|
394
|
+
|
|
395
|
+
for identifier, expected, found, description in (
|
|
396
|
+
(
|
|
397
|
+
"linkage.manifest_campaign",
|
|
398
|
+
manifest.campaign_spec_digest,
|
|
399
|
+
campaign_digest,
|
|
400
|
+
"the bundle manifest names the Campaign it carries",
|
|
401
|
+
),
|
|
402
|
+
(
|
|
403
|
+
"linkage.report_campaign",
|
|
404
|
+
report.campaign_spec_digest,
|
|
405
|
+
campaign_digest,
|
|
406
|
+
"the report was produced under the Campaign the bundle carries",
|
|
407
|
+
),
|
|
408
|
+
(
|
|
409
|
+
"linkage.campaign_policy",
|
|
410
|
+
documents.campaign.data_policy_digest,
|
|
411
|
+
policy_digest,
|
|
412
|
+
"the Campaign names the DataPolicy the bundle carries",
|
|
413
|
+
),
|
|
414
|
+
(
|
|
415
|
+
"linkage.report_policy",
|
|
416
|
+
report.data_policy_digest,
|
|
417
|
+
policy_digest,
|
|
418
|
+
"the report was produced under that same DataPolicy",
|
|
419
|
+
),
|
|
420
|
+
(
|
|
421
|
+
"linkage.manifest_policy",
|
|
422
|
+
manifest.data_policy_digest,
|
|
423
|
+
policy_digest,
|
|
424
|
+
"the bundle manifest names that same DataPolicy",
|
|
425
|
+
),
|
|
426
|
+
(
|
|
427
|
+
"linkage.validation_lock",
|
|
428
|
+
documents.validation_receipt.taskset_lock_digest,
|
|
429
|
+
lock_digest,
|
|
430
|
+
"the validation receipt validates the TasksetLock the bundle carries",
|
|
431
|
+
),
|
|
432
|
+
(
|
|
433
|
+
"linkage.campaign_validation",
|
|
434
|
+
documents.campaign.taskset.validation_receipt_digest,
|
|
435
|
+
receipt_digest,
|
|
436
|
+
"the Campaign commits to that validation receipt",
|
|
437
|
+
),
|
|
438
|
+
(
|
|
439
|
+
"linkage.report_validation",
|
|
440
|
+
report.taskset_validation_receipt_digest,
|
|
441
|
+
receipt_digest,
|
|
442
|
+
"the report cites that validation receipt",
|
|
443
|
+
),
|
|
444
|
+
(
|
|
445
|
+
"linkage.baseline_manifest",
|
|
446
|
+
report.baseline_manifest_digest,
|
|
447
|
+
digest_object(documents.experiments[ExperimentVariant.BASELINE]),
|
|
448
|
+
"the report cites the baseline experiment the bundle carries",
|
|
449
|
+
),
|
|
450
|
+
(
|
|
451
|
+
"linkage.candidate_manifest",
|
|
452
|
+
report.candidate_manifest_digest,
|
|
453
|
+
digest_object(documents.experiments[ExperimentVariant.CANDIDATE]),
|
|
454
|
+
"the report cites the candidate experiment the bundle carries",
|
|
455
|
+
),
|
|
456
|
+
):
|
|
457
|
+
checks.record(
|
|
458
|
+
identifier,
|
|
459
|
+
_PASSED if expected == found else _FAILED,
|
|
460
|
+
COMPARISON_INVALID,
|
|
461
|
+
description
|
|
462
|
+
if expected == found
|
|
463
|
+
else f"{description} — but it names {expected} and this is {found}",
|
|
464
|
+
)
|
|
465
|
+
|
|
466
|
+
committed = list(documents.campaign.taskset.membership.ordered_task_hashes)
|
|
467
|
+
locked = list(documents.taskset_lock.ordered_task_hashes)
|
|
468
|
+
checks.record(
|
|
469
|
+
"linkage.taskset_membership",
|
|
470
|
+
_PASSED if committed == locked else _FAILED,
|
|
471
|
+
COMPARISON_INVALID,
|
|
472
|
+
f"the lock holds the {len(committed)} tasks the Campaign commits to"
|
|
473
|
+
if committed == locked
|
|
474
|
+
else "the lock does not hold the tasks the Campaign commits to",
|
|
475
|
+
)
|
|
476
|
+
checks.record(
|
|
477
|
+
"linkage.run_id",
|
|
478
|
+
_PASSED if manifest.run_id == report.run_id else _FAILED,
|
|
479
|
+
PROOF_BUNDLE_INVALID,
|
|
480
|
+
f"the bundle and the report describe run {report.run_id}"
|
|
481
|
+
if manifest.run_id == report.run_id
|
|
482
|
+
else "the bundle and the report describe different runs",
|
|
483
|
+
)
|
|
484
|
+
|
|
485
|
+
|
|
486
|
+
def _check_receipts(
|
|
487
|
+
directory: Path,
|
|
488
|
+
documents: _Documents,
|
|
489
|
+
identity: ExecutorIdentity,
|
|
490
|
+
checks: _Checks,
|
|
491
|
+
) -> dict[ExperimentVariant, list[EpisodeReceipt]] | None:
|
|
492
|
+
"""Verify every receipt's signature and every variant's commitment."""
|
|
493
|
+
committed = list(documents.taskset_lock.ordered_task_hashes)
|
|
494
|
+
loaded: dict[ExperimentVariant, list[EpisodeReceipt]] = {}
|
|
495
|
+
|
|
496
|
+
for variant in _VARIANT_ORDER:
|
|
497
|
+
receipt_set = documents.receipt_sets[variant]
|
|
498
|
+
envelopes: list[ObjectEnvelope[EpisodeReceipt]] = []
|
|
499
|
+
for position in range(receipt_set.receipt_count):
|
|
500
|
+
relative_path = receipt_filename(variant, position)
|
|
501
|
+
envelope = _load_envelope(
|
|
502
|
+
directory / relative_path, EpisodeReceipt, checks, relative_path
|
|
503
|
+
)
|
|
504
|
+
if envelope is None:
|
|
505
|
+
return None
|
|
506
|
+
envelopes.append(envelope)
|
|
507
|
+
checks.extend(
|
|
508
|
+
verify_signed_object(
|
|
509
|
+
identity=identity, envelope=envelope, subject=relative_path
|
|
510
|
+
).messages
|
|
511
|
+
)
|
|
512
|
+
|
|
513
|
+
try:
|
|
514
|
+
verify_receipt_set(
|
|
515
|
+
manifest=receipt_set,
|
|
516
|
+
signed_receipts=envelopes,
|
|
517
|
+
ordered_task_hashes=committed,
|
|
518
|
+
)
|
|
519
|
+
except VerificationError as error:
|
|
520
|
+
checks.record(
|
|
521
|
+
f"receipt_set.{variant.value}",
|
|
522
|
+
_FAILED,
|
|
523
|
+
RECEIPT_SET_INVALID,
|
|
524
|
+
error.message,
|
|
525
|
+
)
|
|
526
|
+
return None
|
|
527
|
+
|
|
528
|
+
checks.record(
|
|
529
|
+
f"receipt_set.{variant.value}",
|
|
530
|
+
_PASSED,
|
|
531
|
+
RECEIPT_SET_INVALID,
|
|
532
|
+
(
|
|
533
|
+
f"the {variant.value} receipt set commits to its "
|
|
534
|
+
f"{receipt_set.receipt_count} receipts in committed task order"
|
|
535
|
+
),
|
|
536
|
+
)
|
|
537
|
+
expected_manifest = digest_object(documents.experiments[variant])
|
|
538
|
+
checks.record(
|
|
539
|
+
f"receipt_set.{variant.value}.experiment",
|
|
540
|
+
_PASSED
|
|
541
|
+
if receipt_set.experiment_manifest_digest == expected_manifest
|
|
542
|
+
else _FAILED,
|
|
543
|
+
RECEIPT_SET_INVALID,
|
|
544
|
+
f"the {variant.value} receipts were scored under the experiment "
|
|
545
|
+
"manifest the bundle carries"
|
|
546
|
+
if receipt_set.experiment_manifest_digest == expected_manifest
|
|
547
|
+
else f"the {variant.value} receipts were scored under a different "
|
|
548
|
+
"experiment manifest than the one the bundle carries",
|
|
549
|
+
)
|
|
550
|
+
loaded[variant] = [envelope.payload for envelope in envelopes]
|
|
551
|
+
|
|
552
|
+
return loaded
|
|
553
|
+
|
|
554
|
+
|
|
555
|
+
def _check_aggregate(
|
|
556
|
+
documents: _Documents,
|
|
557
|
+
receipts: dict[ExperimentVariant, list[EpisodeReceipt]],
|
|
558
|
+
checks: _Checks,
|
|
559
|
+
) -> None:
|
|
560
|
+
"""Recompute the paired aggregate and require the report to equal it."""
|
|
561
|
+
report = documents.report.payload
|
|
562
|
+
reward = documents.campaign.scoring.primary_reward
|
|
563
|
+
try:
|
|
564
|
+
deltas = pair_task_rewards(
|
|
565
|
+
baseline_receipts=receipts[ExperimentVariant.BASELINE],
|
|
566
|
+
candidate_receipts=receipts[ExperimentVariant.CANDIDATE],
|
|
567
|
+
ordered_task_hashes=list(documents.taskset_lock.ordered_task_hashes),
|
|
568
|
+
reward_name=reward,
|
|
569
|
+
)
|
|
570
|
+
primary = aggregate_primary_result(deltas, reward)
|
|
571
|
+
except VerificationError as error:
|
|
572
|
+
checks.record(
|
|
573
|
+
"aggregate.recomputed",
|
|
574
|
+
_FAILED,
|
|
575
|
+
COMPARISON_INVALID,
|
|
576
|
+
f"the receipts cannot be paired into a comparison: {error.message}",
|
|
577
|
+
)
|
|
578
|
+
return
|
|
579
|
+
|
|
580
|
+
matches = list(deltas) == list(report.task_deltas) and primary == (
|
|
581
|
+
report.primary_result
|
|
582
|
+
)
|
|
583
|
+
checks.record(
|
|
584
|
+
"aggregate.recomputed",
|
|
585
|
+
_PASSED if matches else _FAILED,
|
|
586
|
+
COMPARISON_INVALID,
|
|
587
|
+
(
|
|
588
|
+
f"the report's result is the one these receipts produce: "
|
|
589
|
+
f"{primary.baseline_mean:.4f} against {primary.candidate_mean:.4f} "
|
|
590
|
+
f"on {reward}"
|
|
591
|
+
)
|
|
592
|
+
if matches
|
|
593
|
+
else (
|
|
594
|
+
"the report states a different result than the one its own receipts produce"
|
|
595
|
+
),
|
|
596
|
+
)
|
|
597
|
+
|
|
598
|
+
|
|
599
|
+
def _check_execution_record(
|
|
600
|
+
directory: Path,
|
|
601
|
+
manifest: LocalProofBundleManifest,
|
|
602
|
+
documents: _Documents,
|
|
603
|
+
identity: ExecutorIdentity,
|
|
604
|
+
checks: _Checks,
|
|
605
|
+
) -> None:
|
|
606
|
+
"""Check the comparison's operational record, when the bundle carries one.
|
|
607
|
+
|
|
608
|
+
Three states, and they mean three different things.
|
|
609
|
+
|
|
610
|
+
*The manifest commits to a record.* Then it is held to the same standard as
|
|
611
|
+
everything else: it parses, its signature verifies against the same key,
|
|
612
|
+
and it describes this run and these two experiments. A record that fails
|
|
613
|
+
any of those is a failed check — an operational claim signed into a proof
|
|
614
|
+
is still a claim.
|
|
615
|
+
|
|
616
|
+
*Nothing is there and nothing was promised.* Decisions document 0007 R6:
|
|
617
|
+
the economics are unknown, the measurement is untouched, and the reader is
|
|
618
|
+
told so as a warning rather than a failure.
|
|
619
|
+
|
|
620
|
+
*A file is there that the manifest never named.* That is a failure. The
|
|
621
|
+
signed index is what binds a record to this run, and bytes that arrived
|
|
622
|
+
outside it are not evidence of anything.
|
|
623
|
+
"""
|
|
624
|
+
committed = manifest.artifact(EXECUTION_RECORD_FILENAME)
|
|
625
|
+
path = directory / EXECUTION_RECORD_FILENAME
|
|
626
|
+
if committed is None:
|
|
627
|
+
present = path.is_file()
|
|
628
|
+
checks.record(
|
|
629
|
+
"execution_record.present",
|
|
630
|
+
_FAILED if present else _WARNING,
|
|
631
|
+
COMPARISON_EXECUTION_RECORD_INVALID
|
|
632
|
+
if present
|
|
633
|
+
else OPERATIONAL_EVIDENCE_UNAVAILABLE,
|
|
634
|
+
(
|
|
635
|
+
"this bundle holds a comparison execution record its signed "
|
|
636
|
+
"manifest does not commit to, so nothing binds it to this run"
|
|
637
|
+
if present
|
|
638
|
+
else (
|
|
639
|
+
"this bundle carries no comparison execution record, so "
|
|
640
|
+
"the cost and timing of this comparison are unavailable; "
|
|
641
|
+
"what it measured is unaffected"
|
|
642
|
+
)
|
|
643
|
+
),
|
|
644
|
+
)
|
|
645
|
+
return
|
|
646
|
+
|
|
647
|
+
envelope = _load_envelope(
|
|
648
|
+
path, ComparisonExecutionRecord, checks, "execution-record"
|
|
649
|
+
)
|
|
650
|
+
if envelope is None:
|
|
651
|
+
return
|
|
652
|
+
checks.extend(
|
|
653
|
+
verify_signed_object(
|
|
654
|
+
identity=identity, envelope=envelope, subject="execution-record"
|
|
655
|
+
).messages
|
|
656
|
+
)
|
|
657
|
+
|
|
658
|
+
record = envelope.payload
|
|
659
|
+
describes_this_run = (
|
|
660
|
+
record.run_id == manifest.run_id
|
|
661
|
+
and record.campaign_spec_digest == manifest.campaign_spec_digest
|
|
662
|
+
)
|
|
663
|
+
checks.record(
|
|
664
|
+
"execution_record.run",
|
|
665
|
+
_PASSED if describes_this_run else _FAILED,
|
|
666
|
+
COMPARISON_EXECUTION_RECORD_INVALID,
|
|
667
|
+
f"the execution record describes run {manifest.run_id}"
|
|
668
|
+
if describes_this_run
|
|
669
|
+
else "the execution record describes a different run or Campaign",
|
|
670
|
+
)
|
|
671
|
+
|
|
672
|
+
expected = {
|
|
673
|
+
variant: digest_object(documents.experiments[variant])
|
|
674
|
+
for variant in _VARIANT_ORDER
|
|
675
|
+
}
|
|
676
|
+
same_experiments = all(
|
|
677
|
+
record.side(variant).experiment_manifest_digest == expected[variant]
|
|
678
|
+
for variant in _VARIANT_ORDER
|
|
679
|
+
)
|
|
680
|
+
checks.record(
|
|
681
|
+
"execution_record.experiments",
|
|
682
|
+
_PASSED if same_experiments else _FAILED,
|
|
683
|
+
COMPARISON_EXECUTION_RECORD_INVALID,
|
|
684
|
+
"the execution record accounts for the two experiments this bundle carries"
|
|
685
|
+
if same_experiments
|
|
686
|
+
else (
|
|
687
|
+
"the execution record accounts for different experiments than the "
|
|
688
|
+
"ones this bundle carries"
|
|
689
|
+
),
|
|
690
|
+
)
|
|
691
|
+
|
|
692
|
+
|
|
693
|
+
def _check_publication(report: UpliftReport, checks: _Checks) -> None:
|
|
694
|
+
"""Require the report's two publication fields to hold together.
|
|
695
|
+
|
|
696
|
+
A bundle is written before anybody has been asked whether to publish the
|
|
697
|
+
run, and it is never rewritten afterwards, so the status inside it is
|
|
698
|
+
always one of the two a fresh report can carry: nobody has asked, or the
|
|
699
|
+
rights forbid it. A bundle claiming its report was already published would
|
|
700
|
+
be a bundle somebody had edited after the fact.
|
|
701
|
+
|
|
702
|
+
Eligibility is the other half and it is recomputed rather than read. The
|
|
703
|
+
flag is a stored field in a signed document, and what it is supposed to be
|
|
704
|
+
follows from two other fields of the same document, so the check is the
|
|
705
|
+
same shape as the aggregate recomputation above it: work it out again from
|
|
706
|
+
the evidence, and compare.
|
|
707
|
+
"""
|
|
708
|
+
status = report.statuses.publication
|
|
709
|
+
unpublished = status in (
|
|
710
|
+
PublicationStatus.NOT_REQUESTED,
|
|
711
|
+
PublicationStatus.BLOCKED,
|
|
712
|
+
)
|
|
713
|
+
checks.record(
|
|
714
|
+
"publication.not_requested",
|
|
715
|
+
_PASSED if unpublished else _FAILED,
|
|
716
|
+
PROOF_BUNDLE_INVALID,
|
|
717
|
+
f"nothing in this proof was published: publication is {status.value}"
|
|
718
|
+
if unpublished
|
|
719
|
+
else (
|
|
720
|
+
f"this report's publication status is {status.value}, and a proof "
|
|
721
|
+
"bundle is written before anything could have been published"
|
|
722
|
+
),
|
|
723
|
+
)
|
|
724
|
+
|
|
725
|
+
# The flag says what the build that wrote this report allowed, and that is
|
|
726
|
+
# a property of that build rather than of the run. Recomputing it under
|
|
727
|
+
# today's rules and demanding agreement fails every report signed before
|
|
728
|
+
# publishing existed — including the certification runs this release rests
|
|
729
|
+
# on — and it protects nothing, because the flag sits inside the signed
|
|
730
|
+
# payload and any edit to it breaks the signature two checks above.
|
|
731
|
+
#
|
|
732
|
+
# What a proof can honestly check is that the report does not overclaim:
|
|
733
|
+
# a report may not say it is publishable when its own grade or its own
|
|
734
|
+
# rights statement forbid it. That is the direction that matters, it is a
|
|
735
|
+
# property of the signed bytes alone, and it holds however the rules move.
|
|
736
|
+
overclaims = report.publication_eligible and not publication_eligible_for(
|
|
737
|
+
grade=report.proof_grade, publication=status
|
|
738
|
+
)
|
|
739
|
+
checks.record(
|
|
740
|
+
"publication.eligibility_not_overclaimed",
|
|
741
|
+
_FAILED if overclaims else _PASSED,
|
|
742
|
+
PROOF_BUNDLE_INVALID,
|
|
743
|
+
(
|
|
744
|
+
f"the report claims it may be published while its grade is "
|
|
745
|
+
f"{report.proof_grade} and its publication status is "
|
|
746
|
+
f"{status.value}, which do not allow it"
|
|
747
|
+
)
|
|
748
|
+
if overclaims
|
|
749
|
+
else (
|
|
750
|
+
"the report claims no more about publishing than its own grade "
|
|
751
|
+
"and rights statement allow"
|
|
752
|
+
),
|
|
753
|
+
)
|
|
754
|
+
|
|
755
|
+
|
|
756
|
+
def _check_p1_conditions(
|
|
757
|
+
*,
|
|
758
|
+
manifest: LocalProofBundleManifest,
|
|
759
|
+
documents: _Documents,
|
|
760
|
+
receipts: dict[ExperimentVariant, list[EpisodeReceipt]] | None,
|
|
761
|
+
identity_matches: bool,
|
|
762
|
+
checks: _Checks,
|
|
763
|
+
) -> None:
|
|
764
|
+
"""Re-derive every decisions-0005 section 3.4 condition from these bytes.
|
|
765
|
+
|
|
766
|
+
A report that claims ``P1`` and cannot re-establish all six is overclaiming
|
|
767
|
+
and each missing condition is a failure. A report that claims nothing
|
|
768
|
+
records the same conditions as warnings, so a reader can see exactly what a
|
|
769
|
+
development-only bundle is missing rather than being told only that it is
|
|
770
|
+
not P1.
|
|
771
|
+
"""
|
|
772
|
+
report = documents.report.payload
|
|
773
|
+
claimed = report.proof_grade == "P1"
|
|
774
|
+
established = {
|
|
775
|
+
P1_ARTIFACT_DIGESTS_VERIFY: not [
|
|
776
|
+
message
|
|
777
|
+
for message in checks.messages
|
|
778
|
+
if message.id.startswith("artifact.") and message.status == _FAILED
|
|
779
|
+
],
|
|
780
|
+
P1_RECEIPTS_SIGNED: receipts is not None
|
|
781
|
+
and not [
|
|
782
|
+
message
|
|
783
|
+
for message in checks.messages
|
|
784
|
+
if message.id.startswith("receipts/") and message.status == _FAILED
|
|
785
|
+
],
|
|
786
|
+
P1_REPORT_SIGNED: documents.report.signature is not None
|
|
787
|
+
and not [
|
|
788
|
+
message
|
|
789
|
+
for message in checks.messages
|
|
790
|
+
if message.id.startswith("uplift-report.") and message.status == _FAILED
|
|
791
|
+
],
|
|
792
|
+
P1_PUBLIC_KEY_PRESENT: identity_matches,
|
|
793
|
+
P1_COMPARISON_CONTROLLED: report.statuses.comparison
|
|
794
|
+
in (ComparisonStatus.CONTROLLED, ComparisonStatus.CONTROLLED_WITH_WARNINGS),
|
|
795
|
+
P1_SCORE_VALID: report.statuses.score.value == "valid",
|
|
796
|
+
}
|
|
797
|
+
for condition, holds in established.items():
|
|
798
|
+
status: VerificationStatus = (
|
|
799
|
+
_PASSED if holds else (_FAILED if claimed else _WARNING)
|
|
800
|
+
)
|
|
801
|
+
checks.record(
|
|
802
|
+
f"p1.{condition}",
|
|
803
|
+
status,
|
|
804
|
+
PROOF_BUNDLE_INVALID,
|
|
805
|
+
_p1_detail(condition, holds=holds, claimed=claimed),
|
|
806
|
+
)
|
|
807
|
+
checks.record(
|
|
808
|
+
"p1.grade",
|
|
809
|
+
_PASSED,
|
|
810
|
+
PROOF_BUNDLE_INVALID,
|
|
811
|
+
(
|
|
812
|
+
f"this report claims proof grade P1, which means {P1_MEANING}"
|
|
813
|
+
if claimed
|
|
814
|
+
else (
|
|
815
|
+
f"this report claims proof grade {report.proof_grade}, which is "
|
|
816
|
+
"not evidence of anything"
|
|
817
|
+
)
|
|
818
|
+
),
|
|
819
|
+
)
|
|
820
|
+
# The manifest is named here so that a reader of the P1 block can see which
|
|
821
|
+
# run it belongs to without scrolling back to the linkage checks.
|
|
822
|
+
checks.record(
|
|
823
|
+
"p1.run",
|
|
824
|
+
_PASSED,
|
|
825
|
+
PROOF_BUNDLE_INVALID,
|
|
826
|
+
f"these conditions were checked for run {manifest.run_id}",
|
|
827
|
+
)
|
|
828
|
+
|
|
829
|
+
|
|
830
|
+
def _p1_detail(condition: str, *, holds: bool, claimed: bool) -> str:
|
|
831
|
+
"""Return the sentence one section 3.4 condition reports itself with."""
|
|
832
|
+
statements = {
|
|
833
|
+
P1_ARTIFACT_DIGESTS_VERIFY: "every referenced artifact digest verifies",
|
|
834
|
+
P1_RECEIPTS_SIGNED: "every EpisodeReceipt travels in a signed envelope",
|
|
835
|
+
P1_REPORT_SIGNED: "the UpliftReport travels in a signed envelope",
|
|
836
|
+
P1_PUBLIC_KEY_PRESENT: "the local public key is included in the bundle",
|
|
837
|
+
P1_COMPARISON_CONTROLLED: "the comparison is controlled",
|
|
838
|
+
P1_SCORE_VALID: "the score status is valid",
|
|
839
|
+
}
|
|
840
|
+
statement = statements[condition]
|
|
841
|
+
if holds:
|
|
842
|
+
return statement
|
|
843
|
+
if claimed:
|
|
844
|
+
return f"this report claims P1 and {statement} does not hold"
|
|
845
|
+
return f"{statement} does not hold, and this report does not claim P1"
|
|
846
|
+
|
|
847
|
+
|
|
848
|
+
# ---------------------------------------------------------------------------
|
|
849
|
+
# The five headings a person reads
|
|
850
|
+
# ---------------------------------------------------------------------------
|
|
851
|
+
|
|
852
|
+
|
|
853
|
+
def _explain(result: VerificationResult) -> list[VerificationMessage]:
|
|
854
|
+
"""Group a verification into the categories spec section 7.12 separates."""
|
|
855
|
+
integrity = _worst(
|
|
856
|
+
result,
|
|
857
|
+
lambda identifier: (
|
|
858
|
+
identifier.startswith(("artifact.", "bundle."))
|
|
859
|
+
or identifier.endswith(
|
|
860
|
+
(".signature", ".payload_digest", ".signature_present")
|
|
861
|
+
)
|
|
862
|
+
),
|
|
863
|
+
)
|
|
864
|
+
science = _worst(
|
|
865
|
+
result,
|
|
866
|
+
lambda identifier: identifier.startswith(
|
|
867
|
+
("linkage.", "aggregate.", "receipt_set.")
|
|
868
|
+
),
|
|
869
|
+
)
|
|
870
|
+
attestation = _worst(
|
|
871
|
+
result, lambda identifier: identifier.startswith(("p1.", "uplift-report."))
|
|
872
|
+
)
|
|
873
|
+
publication = _worst(
|
|
874
|
+
result, lambda identifier: identifier.startswith("publication.")
|
|
875
|
+
)
|
|
876
|
+
|
|
877
|
+
return [
|
|
878
|
+
VerificationMessage(
|
|
879
|
+
id="integrity",
|
|
880
|
+
status=integrity,
|
|
881
|
+
code=SIGNATURE_VERIFICATION_FAILED,
|
|
882
|
+
detail=(
|
|
883
|
+
"Cryptographic integrity: every file still matches the digest "
|
|
884
|
+
"it was committed under, and every signature verifies."
|
|
885
|
+
if integrity == _PASSED
|
|
886
|
+
else "Cryptographic integrity: something in this proof no "
|
|
887
|
+
"longer matches what was signed."
|
|
888
|
+
),
|
|
889
|
+
),
|
|
890
|
+
VerificationMessage(
|
|
891
|
+
id="comparison_validity",
|
|
892
|
+
status=science,
|
|
893
|
+
code=COMPARISON_INVALID,
|
|
894
|
+
detail=(
|
|
895
|
+
"Scientific comparison: the documents describe one controlled "
|
|
896
|
+
"comparison, and the report's numbers are the ones its own "
|
|
897
|
+
"receipts produce."
|
|
898
|
+
if science == _PASSED
|
|
899
|
+
else "Scientific comparison: these documents do not describe "
|
|
900
|
+
"one consistent comparison."
|
|
901
|
+
),
|
|
902
|
+
),
|
|
903
|
+
VerificationMessage(
|
|
904
|
+
id="participant_attestation",
|
|
905
|
+
status=attestation,
|
|
906
|
+
code=LOCAL_IDENTITY_INVALID,
|
|
907
|
+
detail=(
|
|
908
|
+
"Participant attestation: signed by the participant's own "
|
|
909
|
+
f"local key. P1 means {P1_MEANING}."
|
|
910
|
+
),
|
|
911
|
+
),
|
|
912
|
+
VerificationMessage(
|
|
913
|
+
id="independent_reproduction",
|
|
914
|
+
status=_WARNING,
|
|
915
|
+
code=PROOF_BUNDLE_INVALID,
|
|
916
|
+
detail=(
|
|
917
|
+
"No independent reproduction: nobody else has run this "
|
|
918
|
+
"comparison, and no platform witnessed it."
|
|
919
|
+
),
|
|
920
|
+
),
|
|
921
|
+
VerificationMessage(
|
|
922
|
+
id="public_publication",
|
|
923
|
+
status=publication,
|
|
924
|
+
code=PROOF_BUNDLE_INVALID,
|
|
925
|
+
detail=(
|
|
926
|
+
"Publication was not requested when this proof was written, "
|
|
927
|
+
"which is the only answer a bundle can give: it is sealed "
|
|
928
|
+
"before anybody could have been asked. It is not a statement "
|
|
929
|
+
"about whether the run was published afterwards. Whether it "
|
|
930
|
+
"may be published is checked separately."
|
|
931
|
+
),
|
|
932
|
+
),
|
|
933
|
+
]
|
|
934
|
+
|
|
935
|
+
|
|
936
|
+
def _worst(
|
|
937
|
+
result: VerificationResult, selector: Callable[[str], bool]
|
|
938
|
+
) -> VerificationStatus:
|
|
939
|
+
"""Return the worst status among the checks a selector matches."""
|
|
940
|
+
statuses = [message.status for message in result.messages if selector(message.id)]
|
|
941
|
+
if not statuses:
|
|
942
|
+
return _FAILED
|
|
943
|
+
if _FAILED in statuses:
|
|
944
|
+
return _FAILED
|
|
945
|
+
if _WARNING in statuses:
|
|
946
|
+
return _WARNING
|
|
947
|
+
return _PASSED
|
|
948
|
+
|
|
949
|
+
|
|
950
|
+
# ---------------------------------------------------------------------------
|
|
951
|
+
# Reading files without letting one bad file stop the report
|
|
952
|
+
# ---------------------------------------------------------------------------
|
|
953
|
+
|
|
954
|
+
|
|
955
|
+
class _Checks:
|
|
956
|
+
"""Collects named checks and turns them into one verdict."""
|
|
957
|
+
|
|
958
|
+
def __init__(self) -> None:
|
|
959
|
+
self.messages: list[VerificationMessage] = []
|
|
960
|
+
|
|
961
|
+
def record(
|
|
962
|
+
self, identifier: str, status: VerificationStatus, code: str, detail: str
|
|
963
|
+
) -> None:
|
|
964
|
+
"""Record one check."""
|
|
965
|
+
self.messages.append(
|
|
966
|
+
VerificationMessage(id=identifier, status=status, code=code, detail=detail)
|
|
967
|
+
)
|
|
968
|
+
|
|
969
|
+
def extend(self, messages: Sequence[VerificationMessage]) -> None:
|
|
970
|
+
"""Record checks another verifier already ran."""
|
|
971
|
+
self.messages.extend(messages)
|
|
972
|
+
|
|
973
|
+
def result(self) -> VerificationResult:
|
|
974
|
+
"""Return the collected verdict."""
|
|
975
|
+
failed = [message for message in self.messages if message.status == _FAILED]
|
|
976
|
+
return VerificationResult(verified=not failed, messages=self.messages)
|
|
977
|
+
|
|
978
|
+
|
|
979
|
+
def _load_model[ModelT: BaseModel](
|
|
980
|
+
path: Path, model: type[ModelT], checks: _Checks
|
|
981
|
+
) -> ModelT | None:
|
|
982
|
+
"""Parse one bundle document from its stored bytes."""
|
|
983
|
+
try:
|
|
984
|
+
raw = path.read_bytes()
|
|
985
|
+
except OSError:
|
|
986
|
+
checks.record(
|
|
987
|
+
f"document.{path.name}",
|
|
988
|
+
_FAILED,
|
|
989
|
+
PROOF_BUNDLE_INVALID,
|
|
990
|
+
f"this bundle has no {path.name}",
|
|
991
|
+
)
|
|
992
|
+
return None
|
|
993
|
+
try:
|
|
994
|
+
return model.model_validate_json(raw)
|
|
995
|
+
except PydanticValidationError as error:
|
|
996
|
+
checks.record(
|
|
997
|
+
f"document.{path.name}",
|
|
998
|
+
_FAILED,
|
|
999
|
+
PROOF_BUNDLE_INVALID,
|
|
1000
|
+
f"{path.name} is not a valid {model.__name__}: {error.errors()[0]['msg']}",
|
|
1001
|
+
)
|
|
1002
|
+
return None
|
|
1003
|
+
|
|
1004
|
+
|
|
1005
|
+
def _load_envelope[ModelT: BaseModel](
|
|
1006
|
+
path: Path, model: type[ModelT], checks: _Checks, subject: str
|
|
1007
|
+
) -> ObjectEnvelope[ModelT] | None:
|
|
1008
|
+
"""Parse one signed envelope from its stored bytes."""
|
|
1009
|
+
try:
|
|
1010
|
+
raw = path.read_bytes()
|
|
1011
|
+
except OSError:
|
|
1012
|
+
checks.record(
|
|
1013
|
+
f"{subject}.present",
|
|
1014
|
+
_FAILED,
|
|
1015
|
+
PROOF_BUNDLE_INVALID,
|
|
1016
|
+
f"this proof has no {path.name}",
|
|
1017
|
+
)
|
|
1018
|
+
return None
|
|
1019
|
+
try:
|
|
1020
|
+
return ObjectEnvelope[model].model_validate_json(raw) # type: ignore[valid-type]
|
|
1021
|
+
except PydanticValidationError as error:
|
|
1022
|
+
checks.record(
|
|
1023
|
+
f"{subject}.present",
|
|
1024
|
+
_FAILED,
|
|
1025
|
+
PROOF_BUNDLE_INVALID,
|
|
1026
|
+
f"{path.name} is not a signed {model.__name__}: {error.errors()[0]['msg']}",
|
|
1027
|
+
)
|
|
1028
|
+
return None
|
|
1029
|
+
|
|
1030
|
+
|
|
1031
|
+
def _load_identity(path: Path, checks: _Checks) -> ExecutorIdentity | None:
|
|
1032
|
+
"""Parse the public identity a proof travels with."""
|
|
1033
|
+
try:
|
|
1034
|
+
raw = path.read_bytes()
|
|
1035
|
+
except OSError:
|
|
1036
|
+
checks.record(
|
|
1037
|
+
"identity.present",
|
|
1038
|
+
_FAILED,
|
|
1039
|
+
LOCAL_IDENTITY_INVALID,
|
|
1040
|
+
(
|
|
1041
|
+
f"this proof carries no {path.name}, so there is no key to "
|
|
1042
|
+
"check its signatures against"
|
|
1043
|
+
),
|
|
1044
|
+
)
|
|
1045
|
+
return None
|
|
1046
|
+
try:
|
|
1047
|
+
return ExecutorIdentity.model_validate_json(raw)
|
|
1048
|
+
except PydanticValidationError as error:
|
|
1049
|
+
checks.record(
|
|
1050
|
+
"identity.present",
|
|
1051
|
+
_FAILED,
|
|
1052
|
+
LOCAL_IDENTITY_INVALID,
|
|
1053
|
+
f"{path.name} is not a valid identity: {error.errors()[0]['msg']}",
|
|
1054
|
+
)
|
|
1055
|
+
return None
|