techtree 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- techtree/__init__.py +35 -0
- techtree/__main__.py +14 -0
- techtree/canonical.py +239 -0
- techtree/catalog/__init__.py +25 -0
- techtree/catalog/repository.py +400 -0
- techtree/catalog/service.py +419 -0
- techtree/cli/__init__.py +1 -0
- techtree/cli/app.py +416 -0
- techtree/cli/commands/__init__.py +1 -0
- techtree/cli/commands/climb.py +1223 -0
- techtree/cli/commands/doctor.py +147 -0
- techtree/cli/commands/engine.py +207 -0
- techtree/cli/commands/proof.py +556 -0
- techtree/cli/commands/publish.py +447 -0
- techtree/cli/commands/release.py +303 -0
- techtree/cli/commands/run.py +1067 -0
- techtree/cli/commands/setup.py +181 -0
- techtree/cli/commands/skill.py +221 -0
- techtree/cli/commands/uplift.py +698 -0
- techtree/cli/commands/withdraw.py +212 -0
- techtree/cli/confirm.py +47 -0
- techtree/cli/context.py +96 -0
- techtree/cli/invoke.py +220 -0
- techtree/cli/output.py +280 -0
- techtree/constants.py +138 -0
- techtree/crypto.py +128 -0
- techtree/doctor/__init__.py +1 -0
- techtree/doctor/checks.py +675 -0
- techtree/doctor/execution_checks.py +435 -0
- techtree/doctor/service.py +326 -0
- techtree/drafts/__init__.py +32 -0
- techtree/drafts/source.py +146 -0
- techtree/drafts/store.py +992 -0
- techtree/engines/__init__.py +1 -0
- techtree/engines/bundle.py +251 -0
- techtree/engines/installer.py +679 -0
- techtree/engines/registry.py +235 -0
- techtree/engines/runner.py +170 -0
- techtree/errors.py +262 -0
- techtree/fs.py +234 -0
- techtree/harness.py +108 -0
- techtree/identity/__init__.py +41 -0
- techtree/identity/models.py +113 -0
- techtree/identity/service.py +199 -0
- techtree/identity/store.py +263 -0
- techtree/ids.py +85 -0
- techtree/manifests/__init__.py +39 -0
- techtree/manifests/builder.py +433 -0
- techtree/manifests/compare.py +376 -0
- techtree/models/__init__.py +282 -0
- techtree/models/base.py +201 -0
- techtree/models/campaign.py +484 -0
- techtree/models/catalog.py +227 -0
- techtree/models/cli.py +151 -0
- techtree/models/climb.py +254 -0
- techtree/models/data_policy.py +130 -0
- techtree/models/engine.py +156 -0
- techtree/models/episode_receipt.py +130 -0
- techtree/models/evaluation_backend.py +113 -0
- techtree/models/experiment.py +154 -0
- techtree/models/run.py +214 -0
- techtree/models/skill.py +156 -0
- techtree/models/uplift_report.py +158 -0
- techtree/models/validation.py +299 -0
- techtree/paths.py +116 -0
- techtree/presentation/__init__.py +31 -0
- techtree/presentation/build.py +1242 -0
- techtree/presentation/compact.py +246 -0
- techtree/presentation/evidence.py +169 -0
- techtree/presentation/models.py +358 -0
- techtree/presentation/rich.py +312 -0
- techtree/presentation/sanitize.py +156 -0
- techtree/publication/__init__.py +44 -0
- techtree/publication/address.py +180 -0
- techtree/publication/coordinates.py +26 -0
- techtree/publication/journal.py +212 -0
- techtree/publication/keccak.py +183 -0
- techtree/publication/models.py +209 -0
- techtree/publication/offer.py +35 -0
- techtree/publication/service.py +618 -0
- techtree/publication/transport.py +296 -0
- techtree/publication/verify.py +242 -0
- techtree/publication/withdraw.py +156 -0
- techtree/py.typed +0 -0
- techtree/receipts/__init__.py +52 -0
- techtree/receipts/bundle.py +578 -0
- techtree/receipts/compare.py +1065 -0
- techtree/receipts/episode.py +672 -0
- techtree/receipts/execution.py +630 -0
- techtree/receipts/observed.py +474 -0
- techtree/receipts/set.py +336 -0
- techtree/receipts/uplift.py +655 -0
- techtree/receipts/verify.py +1055 -0
- techtree/release/__init__.py +9 -0
- techtree/release/bootstrap.py +509 -0
- techtree/release/checks.py +376 -0
- techtree/release/document.py +125 -0
- techtree/release/generate.py +221 -0
- techtree/release/models.py +293 -0
- techtree/release/provenance.py +109 -0
- techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
- techtree/resources/catalog/catalog.json +32 -0
- techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
- techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
- techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
- techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
- techtree/resources/engines/default/engine.json +20 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
- techtree/resources/engines/default/pyproject.toml +23 -0
- techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
- techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
- techtree/resources/engines/default/tools/normalize_validation.py +222 -0
- techtree/resources/engines/default/uv.lock +1758 -0
- techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
- techtree/resources/release/build-provenance.json +4 -0
- techtree/resources/release/release-core.json +24 -0
- techtree/runs/__init__.py +31 -0
- techtree/runs/artifacts.py +750 -0
- techtree/runs/child_registry.py +228 -0
- techtree/runs/events.py +478 -0
- techtree/runs/executor.py +140 -0
- techtree/runs/fake.py +741 -0
- techtree/runs/launcher.py +253 -0
- techtree/runs/machine.py +489 -0
- techtree/runs/real.py +789 -0
- techtree/runs/service.py +616 -0
- techtree/runs/store.py +555 -0
- techtree/runs/validation.py +259 -0
- techtree/runs/variants.py +684 -0
- techtree/settings.py +143 -0
- techtree/skills/__init__.py +14 -0
- techtree/skills/archive.py +282 -0
- techtree/skills/policy.py +62 -0
- techtree/skills/scanner.py +394 -0
- techtree/skills/service.py +752 -0
- techtree/skills/starter.py +434 -0
- techtree/tasksets/__init__.py +1 -0
- techtree/tasksets/membership.py +269 -0
- techtree/tasksets/provider.py +207 -0
- techtree/tasksets/resolver.py +311 -0
- techtree/tasksets/service.py +484 -0
- techtree/tasksets/verifiers_cli.py +538 -0
- techtree/uplift/__init__.py +20 -0
- techtree/uplift/context.py +544 -0
- techtree/uplift/derive.py +203 -0
- techtree/uplift/public_tasks.py +151 -0
- techtree/uplift/service.py +719 -0
- techtree/uplift/source.py +160 -0
- techtree/verifiers/__init__.py +31 -0
- techtree/verifiers/budget.py +219 -0
- techtree/verifiers/child.py +633 -0
- techtree/verifiers/compiler.py +432 -0
- techtree/verifiers/config.py +365 -0
- techtree/verifiers/credentials.py +321 -0
- techtree/verifiers/image.py +126 -0
- techtree/verifiers/models.py +527 -0
- techtree/verifiers/outputs.py +368 -0
- techtree/verifiers/progress.py +192 -0
- techtree/verifiers/supervisor.py +341 -0
- techtree/verifiers/verify.py +782 -0
- techtree/version.py +39 -0
- techtree/worker/__init__.py +18 -0
- techtree/worker/execute.py +487 -0
- techtree/worker/main.py +57 -0
- techtree-0.1.0.dist-info/METADATA +344 -0
- techtree-0.1.0.dist-info/RECORD +174 -0
- techtree-0.1.0.dist-info/WHEEL +4 -0
- techtree-0.1.0.dist-info/entry_points.txt +3 -0
- techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,719 @@
|
|
|
1
|
+
"""The stage that closes a real run. Spec sections 7.9-7.10, 7.20.
|
|
2
|
+
|
|
3
|
+
The real executor stops the moment the evidence is complete: two variants
|
|
4
|
+
executed, their raw output retained, the engine's own normalized projection
|
|
5
|
+
written beside it, and a
|
|
6
|
+
:class:`~techtree.verifiers.models.RealExecutionResult` handed back. Spec
|
|
7
|
+
section 6.22 fixes that boundary deliberately — WP6 does not invent final
|
|
8
|
+
uplift — and until this module existed a real run therefore could not finish at
|
|
9
|
+
all. It is what turns the evidence into the run's result.
|
|
10
|
+
|
|
11
|
+
Everything it does is a pure function of files the run already owns, so it
|
|
12
|
+
spends nothing, starts nothing, and reaches nothing outside the run directory:
|
|
13
|
+
|
|
14
|
+
1. ``building_receipts`` — one receipt per committed task per variant, written
|
|
15
|
+
into the run's receipt tree, signed under the local executor identity, and
|
|
16
|
+
committed to by an ordered receipt-set manifest per variant.
|
|
17
|
+
2. ``verifying_comparison`` — one observed fingerprint per variant, taken from
|
|
18
|
+
the traces and the configuration the engine resolved, checked against the
|
|
19
|
+
manifests and against each other.
|
|
20
|
+
3. ``building_report`` — the paired rewards, the aggregate, the report, its
|
|
21
|
+
signature, and the portable proof bundle the report is verified from before
|
|
22
|
+
it is written through the run store, so that the journal announces the
|
|
23
|
+
digest of a report whose proof already checked out.
|
|
24
|
+
|
|
25
|
+
Four choices are worth stating.
|
|
26
|
+
|
|
27
|
+
*The run's own copies are the only inputs.* The Campaign, the two manifests and
|
|
28
|
+
the committed membership come from ``inputs/``, which the artifact store
|
|
29
|
+
verified against the run's immutable request; the evidence comes from
|
|
30
|
+
``verifiers/<variant>/run/``. Nothing is re-resolved from a draft, a catalog or
|
|
31
|
+
a settings file, so a run's result cannot change because something outside it
|
|
32
|
+
did.
|
|
33
|
+
|
|
34
|
+
*The receipts are written before the comparison is checked.* They are the
|
|
35
|
+
expensive evaluation's only durable scientific record, and a comparison that
|
|
36
|
+
fails is exactly the case where an operator most needs to read them.
|
|
37
|
+
|
|
38
|
+
*The proof is verified before the result is recorded.* Signing is not a
|
|
39
|
+
formality performed on the way out: the bundle is written, verified offline by
|
|
40
|
+
the same code a stranger would run, and only then does the report become the
|
|
41
|
+
run's result. A run whose proof does not check out fails with its evidence
|
|
42
|
+
intact rather than completing with a report nobody could verify.
|
|
43
|
+
|
|
44
|
+
*It records the run's completion itself.* The fake executor writes its own
|
|
45
|
+
result and appends its own completion event, and a real run needs the same two
|
|
46
|
+
acts performed by whoever built the report. Doing it here keeps
|
|
47
|
+
:func:`techtree.worker.execute.execute_run`'s contract — the report it verifies
|
|
48
|
+
against the journal is the report that was recorded — identical for both
|
|
49
|
+
executors.
|
|
50
|
+
|
|
51
|
+
Spec section 7.20's :class:`UpliftService` lives beside it, at the bottom of
|
|
52
|
+
this module. It is a different object with a different job: the report service
|
|
53
|
+
*closes* a run, and the uplift service is what the operator does with a run
|
|
54
|
+
that has already closed — export a sanitized context for a host agent to read,
|
|
55
|
+
hand over the verified text of the Skill that run measured, or prepare the
|
|
56
|
+
Skill-against-Skill comparison that follows from it. Nothing it does starts,
|
|
57
|
+
scores, or signs anything.
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
from __future__ import annotations
|
|
61
|
+
|
|
62
|
+
from collections.abc import Callable
|
|
63
|
+
from dataclasses import dataclass
|
|
64
|
+
from datetime import UTC, datetime
|
|
65
|
+
from pathlib import Path
|
|
66
|
+
from typing import Final
|
|
67
|
+
|
|
68
|
+
from pydantic import ValidationError as PydanticValidationError
|
|
69
|
+
|
|
70
|
+
from techtree.canonical import digest_object
|
|
71
|
+
from techtree.errors import PolicyError, ValidationError, VerificationError
|
|
72
|
+
from techtree.identity.models import ExecutorIdentity
|
|
73
|
+
from techtree.identity.service import IdentityService
|
|
74
|
+
from techtree.models.base import ObjectEnvelope
|
|
75
|
+
from techtree.models.episode_receipt import EpisodeReceipt
|
|
76
|
+
from techtree.models.experiment import ExperimentVariant
|
|
77
|
+
from techtree.models.run import RunPhase, RunRequest
|
|
78
|
+
from techtree.models.uplift_report import UpliftReport
|
|
79
|
+
from techtree.models.validation import TasksetLock
|
|
80
|
+
from techtree.paths import TechtreePaths
|
|
81
|
+
from techtree.receipts.bundle import (
|
|
82
|
+
PROOF_BUNDLE_INVALID,
|
|
83
|
+
LocalProofBundleContents,
|
|
84
|
+
ReferencedObject,
|
|
85
|
+
assess_local_attestation,
|
|
86
|
+
proof_bundle_dir,
|
|
87
|
+
write_local_bundle,
|
|
88
|
+
)
|
|
89
|
+
from techtree.receipts.compare import (
|
|
90
|
+
COMPARISON_INVALID,
|
|
91
|
+
ObservedVariant,
|
|
92
|
+
RealComparisonResult,
|
|
93
|
+
compare_real_variants,
|
|
94
|
+
observe_variant,
|
|
95
|
+
)
|
|
96
|
+
from techtree.receipts.episode import build_variant_receipts, experiment_variant_of
|
|
97
|
+
from techtree.receipts.execution import (
|
|
98
|
+
ComparisonExecutionRecord,
|
|
99
|
+
build_comparison_execution_record,
|
|
100
|
+
)
|
|
101
|
+
from techtree.receipts.observed import read_resolved_config
|
|
102
|
+
from techtree.receipts.set import (
|
|
103
|
+
ReceiptSetManifest,
|
|
104
|
+
build_receipt_set,
|
|
105
|
+
receipt_set_path,
|
|
106
|
+
write_receipt_set,
|
|
107
|
+
)
|
|
108
|
+
from techtree.receipts.uplift import (
|
|
109
|
+
aggregate_primary_result,
|
|
110
|
+
build_uplift_report,
|
|
111
|
+
pair_task_rewards,
|
|
112
|
+
summarize_receipts,
|
|
113
|
+
)
|
|
114
|
+
from techtree.receipts.verify import verify_local_bundle
|
|
115
|
+
from techtree.runs.artifacts import RunArtifactStore, RunInputBundle
|
|
116
|
+
from techtree.runs.events import DETAIL_RESULT_DIGEST, RUN_COMPLETED
|
|
117
|
+
from techtree.runs.executor import raise_if_cancel_requested
|
|
118
|
+
from techtree.runs.real import TASKSET_LOCK_FILENAME
|
|
119
|
+
from techtree.runs.service import RunService
|
|
120
|
+
from techtree.runs.store import RunStore
|
|
121
|
+
from techtree.skills.service import PreparedDraft, SkillPreparationService
|
|
122
|
+
from techtree.uplift.context import (
|
|
123
|
+
SkillImprovementContext,
|
|
124
|
+
build_improvement_context,
|
|
125
|
+
)
|
|
126
|
+
from techtree.uplift.public_tasks import public_projection_for
|
|
127
|
+
from techtree.uplift.source import VerifiedSourceSkill, read_verified_source_skill
|
|
128
|
+
from techtree.verifiers.models import (
|
|
129
|
+
RealExecutionResult,
|
|
130
|
+
RunPaths,
|
|
131
|
+
VariantExecutionResult,
|
|
132
|
+
VariantName,
|
|
133
|
+
)
|
|
134
|
+
from techtree.verifiers.outputs import RESOLVED_CONFIG_PATH
|
|
135
|
+
|
|
136
|
+
__all__ = [
|
|
137
|
+
"REAL_REPORT_STAGE_FAILED",
|
|
138
|
+
"SOURCE_RUN_NOT_USABLE",
|
|
139
|
+
"CompletedRun",
|
|
140
|
+
"RealUpliftReportService",
|
|
141
|
+
"UpliftService",
|
|
142
|
+
"VariantReceipts",
|
|
143
|
+
]
|
|
144
|
+
|
|
145
|
+
#: Stable error code for an input this stage cannot read. The scientific
|
|
146
|
+
#: refusals report through spec section 15's own codes, raised by the modules
|
|
147
|
+
#: that own them.
|
|
148
|
+
REAL_REPORT_STAGE_FAILED: Final = "real_report_stage_failed"
|
|
149
|
+
|
|
150
|
+
#: A finished run that nothing may be derived from. Spec section 7.20's safety
|
|
151
|
+
#: rules all report through this one code, because they answer one question.
|
|
152
|
+
SOURCE_RUN_NOT_USABLE: Final = "source_run_not_usable"
|
|
153
|
+
|
|
154
|
+
#: Both sides, always in comparison order.
|
|
155
|
+
_VARIANT_ORDER: Final[tuple[VariantName, ...]] = (
|
|
156
|
+
VariantName.BASELINE,
|
|
157
|
+
VariantName.CANDIDATE,
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
@dataclass(frozen=True)
|
|
162
|
+
class CompletedRun:
|
|
163
|
+
"""One finished run's signed result and the inputs it was executed from."""
|
|
164
|
+
|
|
165
|
+
report: UpliftReport
|
|
166
|
+
inputs: RunInputBundle
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
@dataclass(frozen=True)
|
|
170
|
+
class VariantReceipts:
|
|
171
|
+
"""One variant's signed receipts, its commitment over them, and what it ran."""
|
|
172
|
+
|
|
173
|
+
receipts: list[EpisodeReceipt]
|
|
174
|
+
signed_receipts: list[ObjectEnvelope[EpisodeReceipt]]
|
|
175
|
+
receipt_set: ReceiptSetManifest
|
|
176
|
+
observed: ObservedVariant
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
class RealUpliftReportService:
|
|
180
|
+
"""Completes a real run by turning its evidence into a signed UpliftReport."""
|
|
181
|
+
|
|
182
|
+
def __init__(
|
|
183
|
+
self,
|
|
184
|
+
*,
|
|
185
|
+
paths: TechtreePaths,
|
|
186
|
+
run_store: RunStore,
|
|
187
|
+
artifact_store: RunArtifactStore,
|
|
188
|
+
identity: IdentityService,
|
|
189
|
+
clock: Callable[[], datetime] | None = None,
|
|
190
|
+
) -> None:
|
|
191
|
+
self._paths = paths
|
|
192
|
+
self._run_store = run_store
|
|
193
|
+
self._artifacts = artifact_store
|
|
194
|
+
self._identity = identity
|
|
195
|
+
self._clock = clock or _utc_now
|
|
196
|
+
|
|
197
|
+
def complete(
|
|
198
|
+
self, *, request: RunRequest, execution: RealExecutionResult
|
|
199
|
+
) -> UpliftReport:
|
|
200
|
+
"""Build, check, aggregate, sign, prove and record this run's result."""
|
|
201
|
+
run_id = request.run_id
|
|
202
|
+
raise_if_cancel_requested(self._run_store, run_id)
|
|
203
|
+
|
|
204
|
+
inputs = self._artifacts.load_inputs(run_id, request)
|
|
205
|
+
run_paths = RunPaths.for_run(self._paths, run_id)
|
|
206
|
+
lock = self._taskset_lock(run_paths, inputs)
|
|
207
|
+
# Before anything is sealed, so that a machine which cannot sign says so
|
|
208
|
+
# while its evidence is still the only thing at stake.
|
|
209
|
+
identity = self._identity.ensure()
|
|
210
|
+
|
|
211
|
+
self._run_store.append(run_id, phase=RunPhase.BUILDING_RECEIPTS)
|
|
212
|
+
sides = {
|
|
213
|
+
variant: self._build_variant(
|
|
214
|
+
request=request,
|
|
215
|
+
inputs=inputs,
|
|
216
|
+
run_paths=run_paths,
|
|
217
|
+
lock=lock,
|
|
218
|
+
variant=variant,
|
|
219
|
+
result=_side(execution, variant),
|
|
220
|
+
)
|
|
221
|
+
for variant in _VARIANT_ORDER
|
|
222
|
+
}
|
|
223
|
+
baseline = sides[VariantName.BASELINE]
|
|
224
|
+
candidate = sides[VariantName.CANDIDATE]
|
|
225
|
+
|
|
226
|
+
self._run_store.append(run_id, phase=RunPhase.VERIFYING_COMPARISON)
|
|
227
|
+
comparison = self._compare(
|
|
228
|
+
inputs=inputs, lock=lock, execution=execution, sides=sides
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
self._run_store.append(run_id, phase=RunPhase.BUILDING_REPORT)
|
|
232
|
+
report = self._report(
|
|
233
|
+
request=request,
|
|
234
|
+
inputs=inputs,
|
|
235
|
+
lock=lock,
|
|
236
|
+
identity=identity,
|
|
237
|
+
comparison=comparison,
|
|
238
|
+
baseline=baseline,
|
|
239
|
+
candidate=candidate,
|
|
240
|
+
)
|
|
241
|
+
# Decisions 0007 R6: what the comparison consumed, recorded beside
|
|
242
|
+
# what it measured and signed with the same key. Built from evidence
|
|
243
|
+
# the run already wrote, so it can neither change the report nor fail
|
|
244
|
+
# the run — a comparison whose economics are unknown is still a
|
|
245
|
+
# comparison, and this record is where that is said out loud.
|
|
246
|
+
execution_record = build_comparison_execution_record(
|
|
247
|
+
run_id=run_id,
|
|
248
|
+
campaign_spec_digest=request.campaign_spec_digest,
|
|
249
|
+
campaign_max_concurrent=inputs.campaign.execution.max_concurrent,
|
|
250
|
+
execution=execution,
|
|
251
|
+
run_root=self._paths.run_dir(run_id),
|
|
252
|
+
)
|
|
253
|
+
self._prove(
|
|
254
|
+
request=request,
|
|
255
|
+
inputs=inputs,
|
|
256
|
+
lock=lock,
|
|
257
|
+
identity=identity,
|
|
258
|
+
report=report,
|
|
259
|
+
baseline=baseline,
|
|
260
|
+
candidate=candidate,
|
|
261
|
+
execution_record=execution_record,
|
|
262
|
+
)
|
|
263
|
+
|
|
264
|
+
self._run_store.write_result(run_id, report)
|
|
265
|
+
self._run_store.append(
|
|
266
|
+
run_id,
|
|
267
|
+
phase=RunPhase.COMPLETED,
|
|
268
|
+
kind=RUN_COMPLETED,
|
|
269
|
+
details={DETAIL_RESULT_DIGEST: digest_object(report)},
|
|
270
|
+
)
|
|
271
|
+
return report
|
|
272
|
+
|
|
273
|
+
# -- receipts -----------------------------------------------------------
|
|
274
|
+
|
|
275
|
+
def _build_variant(
|
|
276
|
+
self,
|
|
277
|
+
*,
|
|
278
|
+
request: RunRequest,
|
|
279
|
+
inputs: RunInputBundle,
|
|
280
|
+
run_paths: RunPaths,
|
|
281
|
+
lock: TasksetLock,
|
|
282
|
+
variant: VariantName,
|
|
283
|
+
result: VariantExecutionResult,
|
|
284
|
+
) -> VariantReceipts:
|
|
285
|
+
"""Build, persist and commit to one variant's receipts."""
|
|
286
|
+
experiment = (
|
|
287
|
+
inputs.baseline if variant is VariantName.BASELINE else inputs.candidate
|
|
288
|
+
)
|
|
289
|
+
campaign = inputs.campaign
|
|
290
|
+
receipts = build_variant_receipts(
|
|
291
|
+
run_request=request,
|
|
292
|
+
variant=variant,
|
|
293
|
+
experiment=experiment,
|
|
294
|
+
result=result,
|
|
295
|
+
evaluation_backend=campaign.evaluation_backend,
|
|
296
|
+
ordered_task_hashes=lock.ordered_task_hashes,
|
|
297
|
+
primary_reward=campaign.scoring.primary_reward,
|
|
298
|
+
evidence=campaign.evidence,
|
|
299
|
+
)
|
|
300
|
+
for position, receipt in enumerate(receipts):
|
|
301
|
+
self._artifacts.write_episode_receipt(
|
|
302
|
+
request.run_id, position=position, receipt=receipt
|
|
303
|
+
)
|
|
304
|
+
|
|
305
|
+
# Signed here, at the moment the receipts exist and before anything is
|
|
306
|
+
# committed to them: the receipt-set manifest commits to the digest of
|
|
307
|
+
# each receipt's payload, which signing does not move, so the
|
|
308
|
+
# commitment and the signature are two independent seals over the same
|
|
309
|
+
# bytes rather than one wrapped in the other.
|
|
310
|
+
envelopes = [self._identity.sign_object(receipt) for receipt in receipts]
|
|
311
|
+
protocol_variant = experiment_variant_of(variant)
|
|
312
|
+
receipt_set = build_receipt_set(
|
|
313
|
+
run_id=request.run_id,
|
|
314
|
+
variant=protocol_variant,
|
|
315
|
+
experiment_manifest_digest=result.experiment_manifest_digest,
|
|
316
|
+
signed_receipts=envelopes,
|
|
317
|
+
ordered_task_hashes=lock.ordered_task_hashes,
|
|
318
|
+
)
|
|
319
|
+
write_receipt_set(
|
|
320
|
+
receipt_set, receipt_set_path(run_paths.root, protocol_variant)
|
|
321
|
+
)
|
|
322
|
+
|
|
323
|
+
return VariantReceipts(
|
|
324
|
+
receipts=receipts,
|
|
325
|
+
signed_receipts=envelopes,
|
|
326
|
+
receipt_set=receipt_set,
|
|
327
|
+
observed=observe_variant(
|
|
328
|
+
result=result,
|
|
329
|
+
resolved_config=read_resolved_config(
|
|
330
|
+
run_paths.variant_output_dir(variant) / RESOLVED_CONFIG_PATH
|
|
331
|
+
),
|
|
332
|
+
runtime=campaign.subject.runtime,
|
|
333
|
+
),
|
|
334
|
+
)
|
|
335
|
+
|
|
336
|
+
# -- comparison ---------------------------------------------------------
|
|
337
|
+
|
|
338
|
+
def _compare(
|
|
339
|
+
self,
|
|
340
|
+
*,
|
|
341
|
+
inputs: RunInputBundle,
|
|
342
|
+
lock: TasksetLock,
|
|
343
|
+
execution: RealExecutionResult,
|
|
344
|
+
sides: dict[VariantName, VariantReceipts],
|
|
345
|
+
) -> RealComparisonResult:
|
|
346
|
+
"""Check that the two executions were one experiment."""
|
|
347
|
+
return compare_real_variants(
|
|
348
|
+
campaign=inputs.campaign,
|
|
349
|
+
baseline_manifest=inputs.baseline,
|
|
350
|
+
candidate_manifest=inputs.candidate,
|
|
351
|
+
prepared_manifest_comparison=inputs.comparison,
|
|
352
|
+
baseline_receipts=sides[VariantName.BASELINE].receipts,
|
|
353
|
+
candidate_receipts=sides[VariantName.CANDIDATE].receipts,
|
|
354
|
+
taskset_lock=lock,
|
|
355
|
+
baseline_observed=sides[VariantName.BASELINE].observed,
|
|
356
|
+
candidate_observed=sides[VariantName.CANDIDATE].observed,
|
|
357
|
+
schedule=execution.schedule,
|
|
358
|
+
)
|
|
359
|
+
|
|
360
|
+
# -- the report ---------------------------------------------------------
|
|
361
|
+
|
|
362
|
+
def _report(
|
|
363
|
+
self,
|
|
364
|
+
*,
|
|
365
|
+
request: RunRequest,
|
|
366
|
+
inputs: RunInputBundle,
|
|
367
|
+
lock: TasksetLock,
|
|
368
|
+
identity: ExecutorIdentity,
|
|
369
|
+
comparison: RealComparisonResult,
|
|
370
|
+
baseline: VariantReceipts,
|
|
371
|
+
candidate: VariantReceipts,
|
|
372
|
+
) -> UpliftReport:
|
|
373
|
+
"""Aggregate the paired rewards and construct the report."""
|
|
374
|
+
campaign = inputs.campaign
|
|
375
|
+
deltas = pair_task_rewards(
|
|
376
|
+
baseline_receipts=baseline.receipts,
|
|
377
|
+
candidate_receipts=candidate.receipts,
|
|
378
|
+
ordered_task_hashes=comparison.ordered_task_hashes,
|
|
379
|
+
reward_name=campaign.scoring.primary_reward,
|
|
380
|
+
)
|
|
381
|
+
primary = aggregate_primary_result(deltas, campaign.scoring.primary_reward)
|
|
382
|
+
score, evidence = summarize_receipts(baseline.receipts, candidate.receipts)
|
|
383
|
+
|
|
384
|
+
return build_uplift_report(
|
|
385
|
+
run_request=request,
|
|
386
|
+
campaign=campaign,
|
|
387
|
+
# The run's own staged copy of the rights statement it executed
|
|
388
|
+
# under, which is what decides whether its report may be published.
|
|
389
|
+
data_policy=inputs.source.data_policy,
|
|
390
|
+
# The Campaign's own commitment. Every validation source in this
|
|
391
|
+
# build is required to have validated against exactly this receipt
|
|
392
|
+
# before a single episode is scored on the taskset, so it is the
|
|
393
|
+
# receipt this run was validated under and not merely a reference.
|
|
394
|
+
taskset_validation_receipt_digest=(
|
|
395
|
+
campaign.taskset.validation_receipt_digest
|
|
396
|
+
),
|
|
397
|
+
baseline_manifest=inputs.baseline,
|
|
398
|
+
candidate_manifest=inputs.candidate,
|
|
399
|
+
baseline_receipt_set=baseline.receipt_set,
|
|
400
|
+
candidate_receipt_set=candidate.receipt_set,
|
|
401
|
+
comparison=comparison,
|
|
402
|
+
task_deltas=deltas,
|
|
403
|
+
primary=primary,
|
|
404
|
+
score=score,
|
|
405
|
+
evidence=evidence,
|
|
406
|
+
# The one argument that decides the grade, and it is decided by the
|
|
407
|
+
# decisions-0005 section 3.4 conditions rather than by the presence
|
|
408
|
+
# of a key. Every condition is checked; any failure withholds the
|
|
409
|
+
# verdict instead of claiming a grade this run did not earn.
|
|
410
|
+
attestation=assess_local_attestation(
|
|
411
|
+
identity=identity,
|
|
412
|
+
identity_self_check=self._identity.store.verify_pair(),
|
|
413
|
+
referenced_objects=self._referenced_objects(
|
|
414
|
+
request=request, inputs=inputs, lock=lock
|
|
415
|
+
),
|
|
416
|
+
signed_receipts={
|
|
417
|
+
ExperimentVariant.BASELINE: baseline.signed_receipts,
|
|
418
|
+
ExperimentVariant.CANDIDATE: candidate.signed_receipts,
|
|
419
|
+
},
|
|
420
|
+
comparison=comparison.status,
|
|
421
|
+
score=score,
|
|
422
|
+
).attestation,
|
|
423
|
+
created_at=self._clock(),
|
|
424
|
+
)
|
|
425
|
+
|
|
426
|
+
# -- the proof ----------------------------------------------------------
|
|
427
|
+
|
|
428
|
+
def _prove(
|
|
429
|
+
self,
|
|
430
|
+
*,
|
|
431
|
+
request: RunRequest,
|
|
432
|
+
inputs: RunInputBundle,
|
|
433
|
+
lock: TasksetLock,
|
|
434
|
+
identity: ExecutorIdentity,
|
|
435
|
+
report: UpliftReport,
|
|
436
|
+
baseline: VariantReceipts,
|
|
437
|
+
candidate: VariantReceipts,
|
|
438
|
+
execution_record: ComparisonExecutionRecord,
|
|
439
|
+
) -> Path:
|
|
440
|
+
"""Sign the report, write the portable proof, and verify it offline.
|
|
441
|
+
|
|
442
|
+
The bundle is verified before the report is recorded, from the bytes
|
|
443
|
+
just written, with the same verifier a stranger would use. That is what
|
|
444
|
+
turns "the report says P1" into "the conditions P1 names hold in this
|
|
445
|
+
directory": a bundle that does not verify fails the run, and no report
|
|
446
|
+
claiming a grade it cannot support becomes a result.
|
|
447
|
+
"""
|
|
448
|
+
signed_report = self._identity.sign_object(report)
|
|
449
|
+
run_root = self._paths.run_dir(request.run_id)
|
|
450
|
+
directory = write_local_bundle(
|
|
451
|
+
run_root=run_root,
|
|
452
|
+
contents=LocalProofBundleContents(
|
|
453
|
+
identity=identity,
|
|
454
|
+
campaign=inputs.campaign,
|
|
455
|
+
data_policy=inputs.source.data_policy,
|
|
456
|
+
taskset_lock=lock,
|
|
457
|
+
validation_receipt=inputs.source.publisher_validation,
|
|
458
|
+
experiments={
|
|
459
|
+
ExperimentVariant.BASELINE: inputs.baseline,
|
|
460
|
+
ExperimentVariant.CANDIDATE: inputs.candidate,
|
|
461
|
+
},
|
|
462
|
+
receipt_sets={
|
|
463
|
+
ExperimentVariant.BASELINE: baseline.receipt_set,
|
|
464
|
+
ExperimentVariant.CANDIDATE: candidate.receipt_set,
|
|
465
|
+
},
|
|
466
|
+
receipts={
|
|
467
|
+
ExperimentVariant.BASELINE: baseline.signed_receipts,
|
|
468
|
+
ExperimentVariant.CANDIDATE: candidate.signed_receipts,
|
|
469
|
+
},
|
|
470
|
+
report=signed_report,
|
|
471
|
+
execution_record=self._identity.sign_object(execution_record),
|
|
472
|
+
),
|
|
473
|
+
identity_service=self._identity,
|
|
474
|
+
)
|
|
475
|
+
|
|
476
|
+
verification = verify_local_bundle(directory)
|
|
477
|
+
if verification.verified:
|
|
478
|
+
return directory
|
|
479
|
+
raise VerificationError(
|
|
480
|
+
"this run's local proof does not verify from the bytes it just "
|
|
481
|
+
"wrote, so its report is not recorded as a result",
|
|
482
|
+
code=PROOF_BUNDLE_INVALID,
|
|
483
|
+
details={
|
|
484
|
+
"run_id": request.run_id,
|
|
485
|
+
"proof_grade": report.proof_grade,
|
|
486
|
+
"failed_checks": [message.id for message in verification.failures],
|
|
487
|
+
},
|
|
488
|
+
)
|
|
489
|
+
|
|
490
|
+
def _referenced_objects(
|
|
491
|
+
self, *, request: RunRequest, inputs: RunInputBundle, lock: TasksetLock
|
|
492
|
+
) -> list[ReferencedObject]:
|
|
493
|
+
"""Return every object this report cites, with the digest it cites it under.
|
|
494
|
+
|
|
495
|
+
Section 3.4's first condition is that all referenced artifact digests
|
|
496
|
+
verify, and this is the list: each object recomputed from itself and
|
|
497
|
+
compared against the digest some *other* document names it by, so no
|
|
498
|
+
value is ever checked against itself.
|
|
499
|
+
"""
|
|
500
|
+
campaign = inputs.campaign
|
|
501
|
+
publisher_validation = inputs.source.publisher_validation
|
|
502
|
+
return [
|
|
503
|
+
ReferencedObject("campaign", campaign, request.campaign_spec_digest),
|
|
504
|
+
ReferencedObject(
|
|
505
|
+
"data-policy",
|
|
506
|
+
inputs.source.data_policy,
|
|
507
|
+
campaign.data_policy_digest,
|
|
508
|
+
),
|
|
509
|
+
ReferencedObject(
|
|
510
|
+
"taskset-validation-receipt",
|
|
511
|
+
publisher_validation,
|
|
512
|
+
campaign.taskset.validation_receipt_digest,
|
|
513
|
+
),
|
|
514
|
+
ReferencedObject(
|
|
515
|
+
"taskset-lock", lock, publisher_validation.taskset_lock_digest
|
|
516
|
+
),
|
|
517
|
+
ReferencedObject(
|
|
518
|
+
"baseline-experiment",
|
|
519
|
+
inputs.baseline,
|
|
520
|
+
request.baseline_manifest_digest,
|
|
521
|
+
),
|
|
522
|
+
ReferencedObject(
|
|
523
|
+
"candidate-experiment",
|
|
524
|
+
inputs.candidate,
|
|
525
|
+
request.candidate_manifest_digest,
|
|
526
|
+
),
|
|
527
|
+
]
|
|
528
|
+
|
|
529
|
+
# -- inputs -------------------------------------------------------------
|
|
530
|
+
|
|
531
|
+
def _taskset_lock(self, run_paths: RunPaths, inputs: RunInputBundle) -> TasksetLock:
|
|
532
|
+
"""Read the run's own copy of the membership its episodes were joined on.
|
|
533
|
+
|
|
534
|
+
Written by the executor before anything was compiled, from the
|
|
535
|
+
validation this run was given, and read back here rather than re-derived
|
|
536
|
+
so that the receipts are ordered by the same list the normalizer used.
|
|
537
|
+
"""
|
|
538
|
+
path = run_paths.inputs_dir / TASKSET_LOCK_FILENAME
|
|
539
|
+
try:
|
|
540
|
+
lock = TasksetLock.model_validate_json(path.read_bytes())
|
|
541
|
+
except OSError as error:
|
|
542
|
+
raise ValidationError(
|
|
543
|
+
"this run recorded no validated taskset lock, so its episodes "
|
|
544
|
+
"cannot be ordered by the membership they were scored under",
|
|
545
|
+
code=REAL_REPORT_STAGE_FAILED,
|
|
546
|
+
details={"run_id": inputs.request.run_id, "path": str(path)},
|
|
547
|
+
) from error
|
|
548
|
+
except PydanticValidationError as error:
|
|
549
|
+
raise ValidationError(
|
|
550
|
+
f"this run's taskset lock cannot be read: {error.errors()[0]['msg']}",
|
|
551
|
+
code=REAL_REPORT_STAGE_FAILED,
|
|
552
|
+
details={"run_id": inputs.request.run_id, "path": str(path)},
|
|
553
|
+
) from error
|
|
554
|
+
|
|
555
|
+
committed = list(inputs.campaign.taskset.membership.ordered_task_hashes)
|
|
556
|
+
if list(lock.ordered_task_hashes) != committed:
|
|
557
|
+
raise ValidationError(
|
|
558
|
+
"this run's taskset lock does not hold the tasks its Campaign "
|
|
559
|
+
"commits to",
|
|
560
|
+
code=COMPARISON_INVALID,
|
|
561
|
+
details={
|
|
562
|
+
"run_id": inputs.request.run_id,
|
|
563
|
+
"locked": len(lock.ordered_task_hashes),
|
|
564
|
+
"committed": len(committed),
|
|
565
|
+
},
|
|
566
|
+
)
|
|
567
|
+
return lock
|
|
568
|
+
|
|
569
|
+
|
|
570
|
+
def _side(
|
|
571
|
+
execution: RealExecutionResult, variant: VariantName
|
|
572
|
+
) -> VariantExecutionResult:
|
|
573
|
+
"""Return one variant's half of the execution result."""
|
|
574
|
+
return (
|
|
575
|
+
execution.baseline if variant is VariantName.BASELINE else execution.candidate
|
|
576
|
+
)
|
|
577
|
+
|
|
578
|
+
|
|
579
|
+
def _utc_now() -> datetime:
|
|
580
|
+
"""Return the current instant in UTC."""
|
|
581
|
+
return datetime.now(UTC)
|
|
582
|
+
|
|
583
|
+
|
|
584
|
+
# ---------------------------------------------------------------------------
|
|
585
|
+
# Spec section 7.20: what an operator does with a run that has finished
|
|
586
|
+
# ---------------------------------------------------------------------------
|
|
587
|
+
|
|
588
|
+
|
|
589
|
+
class UpliftService:
|
|
590
|
+
"""Exports improvement context and prepares Skill-replacement drafts.
|
|
591
|
+
|
|
592
|
+
Both operations read a run that already completed and neither changes it.
|
|
593
|
+
The safety rules in spec section 7.20 are all checked in one place,
|
|
594
|
+
:meth:`_completed_real_run`, because both operations rest on the same
|
|
595
|
+
claim: that this run really executed, really finished, and really verifies
|
|
596
|
+
from its own bytes. A context exported from a development-only run would
|
|
597
|
+
describe invented numbers, and a replacement prepared from one would
|
|
598
|
+
declare a baseline nothing measured.
|
|
599
|
+
"""
|
|
600
|
+
|
|
601
|
+
def __init__(
|
|
602
|
+
self,
|
|
603
|
+
*,
|
|
604
|
+
paths: TechtreePaths,
|
|
605
|
+
run_service: RunService,
|
|
606
|
+
artifact_store: RunArtifactStore,
|
|
607
|
+
skill_service: SkillPreparationService,
|
|
608
|
+
) -> None:
|
|
609
|
+
self._paths = paths
|
|
610
|
+
self._runs = run_service
|
|
611
|
+
self._artifacts = artifact_store
|
|
612
|
+
self._skills = skill_service
|
|
613
|
+
|
|
614
|
+
def improvement_context(self, run_id: str) -> SkillImprovementContext:
|
|
615
|
+
"""Build sanitized local context from a completed real run."""
|
|
616
|
+
source = self._completed_real_run(run_id)
|
|
617
|
+
return build_improvement_context(
|
|
618
|
+
report=source.report,
|
|
619
|
+
candidate_receipts=self._artifacts.episode_receipts(
|
|
620
|
+
run_id, ExperimentVariant.CANDIDATE
|
|
621
|
+
),
|
|
622
|
+
baseline_receipts=self._artifacts.episode_receipts(
|
|
623
|
+
run_id, ExperimentVariant.BASELINE
|
|
624
|
+
),
|
|
625
|
+
campaign=source.inputs.campaign,
|
|
626
|
+
parent_skill=source.inputs.candidate_skill.artifact,
|
|
627
|
+
task_public_projection=public_projection_for(source.inputs.campaign),
|
|
628
|
+
)
|
|
629
|
+
|
|
630
|
+
def verified_source_skill(self, run_id: str) -> VerifiedSourceSkill:
|
|
631
|
+
"""Return the verified text of the Skill this run measured.
|
|
632
|
+
|
|
633
|
+
The same run precondition the context rests on applies here, for the
|
|
634
|
+
same reason: text taken from a run that did not really execute would
|
|
635
|
+
be presented as the Skill a result was measured with, and no result
|
|
636
|
+
like that exists. What is returned is verified twice over — once when
|
|
637
|
+
the run's inputs are loaded, and again on the exact bytes handed back.
|
|
638
|
+
"""
|
|
639
|
+
source = self._completed_real_run(run_id)
|
|
640
|
+
return read_verified_source_skill(source.inputs.candidate_skill, run_id=run_id)
|
|
641
|
+
|
|
642
|
+
def prepare_replacement(
|
|
643
|
+
self,
|
|
644
|
+
*,
|
|
645
|
+
source_run_id: str,
|
|
646
|
+
candidate_skill_path: Path,
|
|
647
|
+
candidate_label: str | None = None,
|
|
648
|
+
) -> PreparedDraft:
|
|
649
|
+
"""Prepare the run that compares the source run's Skill against a new one.
|
|
650
|
+
|
|
651
|
+
The baseline is the source run's *candidate* Skill, taken from that
|
|
652
|
+
run's own staged inputs — which the artifact store re-verifies file by
|
|
653
|
+
file against the artifact before handing them over — rather than
|
|
654
|
+
rescanned from wherever the participant originally wrote it. That is
|
|
655
|
+
what makes the second comparison's baseline the Skill the first
|
|
656
|
+
comparison actually measured, and not a directory that has moved on
|
|
657
|
+
since.
|
|
658
|
+
"""
|
|
659
|
+
source = self._completed_real_run(source_run_id)
|
|
660
|
+
return self._skills.prepare_replacement(
|
|
661
|
+
source_campaign=source.inputs.campaign,
|
|
662
|
+
data_policy=source.inputs.source.data_policy,
|
|
663
|
+
publisher_validation=source.inputs.source.publisher_validation,
|
|
664
|
+
validation_evidence=source.inputs.validation_evidence,
|
|
665
|
+
source_report=source.report,
|
|
666
|
+
baseline_skill=source.inputs.candidate_skill,
|
|
667
|
+
candidate_skill_path=candidate_skill_path,
|
|
668
|
+
candidate_label=candidate_label,
|
|
669
|
+
)
|
|
670
|
+
|
|
671
|
+
# -- the one precondition both operations rest on -----------------------
|
|
672
|
+
|
|
673
|
+
def _completed_real_run(self, run_id: str) -> CompletedRun:
|
|
674
|
+
"""Load a run that finished, executed for real, and verifies offline."""
|
|
675
|
+
# Raises with the run's own phase when it has not finished, and
|
|
676
|
+
# re-derives the report's digest against the journal that announced it.
|
|
677
|
+
report = self._runs.result(run_id)
|
|
678
|
+
inputs = self._artifacts.load_inputs(run_id, self._runs.request(run_id))
|
|
679
|
+
|
|
680
|
+
if report.proof_grade == "development_only":
|
|
681
|
+
raise PolicyError(
|
|
682
|
+
f"run {run_id} was executed by the development executor, so "
|
|
683
|
+
"its numbers are not evidence and nothing may be derived from "
|
|
684
|
+
"them",
|
|
685
|
+
code=SOURCE_RUN_NOT_USABLE,
|
|
686
|
+
details={"run_id": run_id, "proof_grade": report.proof_grade},
|
|
687
|
+
)
|
|
688
|
+
|
|
689
|
+
backends = {
|
|
690
|
+
receipt.execution_backend
|
|
691
|
+
for variant in ExperimentVariant
|
|
692
|
+
for receipt in self._artifacts.episode_receipts(run_id, variant)
|
|
693
|
+
}
|
|
694
|
+
if backends != {"verifiers"}:
|
|
695
|
+
raise PolicyError(
|
|
696
|
+
f"run {run_id} did not evaluate every episode for real, so it "
|
|
697
|
+
"is not a run another comparison can be built on",
|
|
698
|
+
code=SOURCE_RUN_NOT_USABLE,
|
|
699
|
+
details={
|
|
700
|
+
"run_id": run_id,
|
|
701
|
+
"execution_backends": ", ".join(sorted(backends)),
|
|
702
|
+
},
|
|
703
|
+
)
|
|
704
|
+
|
|
705
|
+
verification = verify_local_bundle(
|
|
706
|
+
proof_bundle_dir(self._paths.run_dir(run_id))
|
|
707
|
+
)
|
|
708
|
+
if not verification.verified:
|
|
709
|
+
raise VerificationError(
|
|
710
|
+
f"run {run_id}'s local proof does not verify, so nothing may "
|
|
711
|
+
"be derived from what it reported",
|
|
712
|
+
code=PROOF_BUNDLE_INVALID,
|
|
713
|
+
details={
|
|
714
|
+
"run_id": run_id,
|
|
715
|
+
"failed_checks": [message.id for message in verification.failures],
|
|
716
|
+
},
|
|
717
|
+
)
|
|
718
|
+
|
|
719
|
+
return CompletedRun(report=report, inputs=inputs)
|