techtree 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- techtree/__init__.py +35 -0
- techtree/__main__.py +14 -0
- techtree/canonical.py +239 -0
- techtree/catalog/__init__.py +25 -0
- techtree/catalog/repository.py +400 -0
- techtree/catalog/service.py +419 -0
- techtree/cli/__init__.py +1 -0
- techtree/cli/app.py +416 -0
- techtree/cli/commands/__init__.py +1 -0
- techtree/cli/commands/climb.py +1223 -0
- techtree/cli/commands/doctor.py +147 -0
- techtree/cli/commands/engine.py +207 -0
- techtree/cli/commands/proof.py +556 -0
- techtree/cli/commands/publish.py +447 -0
- techtree/cli/commands/release.py +303 -0
- techtree/cli/commands/run.py +1067 -0
- techtree/cli/commands/setup.py +181 -0
- techtree/cli/commands/skill.py +221 -0
- techtree/cli/commands/uplift.py +698 -0
- techtree/cli/commands/withdraw.py +212 -0
- techtree/cli/confirm.py +47 -0
- techtree/cli/context.py +96 -0
- techtree/cli/invoke.py +220 -0
- techtree/cli/output.py +280 -0
- techtree/constants.py +138 -0
- techtree/crypto.py +128 -0
- techtree/doctor/__init__.py +1 -0
- techtree/doctor/checks.py +675 -0
- techtree/doctor/execution_checks.py +435 -0
- techtree/doctor/service.py +326 -0
- techtree/drafts/__init__.py +32 -0
- techtree/drafts/source.py +146 -0
- techtree/drafts/store.py +992 -0
- techtree/engines/__init__.py +1 -0
- techtree/engines/bundle.py +251 -0
- techtree/engines/installer.py +679 -0
- techtree/engines/registry.py +235 -0
- techtree/engines/runner.py +170 -0
- techtree/errors.py +262 -0
- techtree/fs.py +234 -0
- techtree/harness.py +108 -0
- techtree/identity/__init__.py +41 -0
- techtree/identity/models.py +113 -0
- techtree/identity/service.py +199 -0
- techtree/identity/store.py +263 -0
- techtree/ids.py +85 -0
- techtree/manifests/__init__.py +39 -0
- techtree/manifests/builder.py +433 -0
- techtree/manifests/compare.py +376 -0
- techtree/models/__init__.py +282 -0
- techtree/models/base.py +201 -0
- techtree/models/campaign.py +484 -0
- techtree/models/catalog.py +227 -0
- techtree/models/cli.py +151 -0
- techtree/models/climb.py +254 -0
- techtree/models/data_policy.py +130 -0
- techtree/models/engine.py +156 -0
- techtree/models/episode_receipt.py +130 -0
- techtree/models/evaluation_backend.py +113 -0
- techtree/models/experiment.py +154 -0
- techtree/models/run.py +214 -0
- techtree/models/skill.py +156 -0
- techtree/models/uplift_report.py +158 -0
- techtree/models/validation.py +299 -0
- techtree/paths.py +116 -0
- techtree/presentation/__init__.py +31 -0
- techtree/presentation/build.py +1242 -0
- techtree/presentation/compact.py +246 -0
- techtree/presentation/evidence.py +169 -0
- techtree/presentation/models.py +358 -0
- techtree/presentation/rich.py +312 -0
- techtree/presentation/sanitize.py +156 -0
- techtree/publication/__init__.py +44 -0
- techtree/publication/address.py +180 -0
- techtree/publication/coordinates.py +26 -0
- techtree/publication/journal.py +212 -0
- techtree/publication/keccak.py +183 -0
- techtree/publication/models.py +209 -0
- techtree/publication/offer.py +35 -0
- techtree/publication/service.py +618 -0
- techtree/publication/transport.py +296 -0
- techtree/publication/verify.py +242 -0
- techtree/publication/withdraw.py +156 -0
- techtree/py.typed +0 -0
- techtree/receipts/__init__.py +52 -0
- techtree/receipts/bundle.py +578 -0
- techtree/receipts/compare.py +1065 -0
- techtree/receipts/episode.py +672 -0
- techtree/receipts/execution.py +630 -0
- techtree/receipts/observed.py +474 -0
- techtree/receipts/set.py +336 -0
- techtree/receipts/uplift.py +655 -0
- techtree/receipts/verify.py +1055 -0
- techtree/release/__init__.py +9 -0
- techtree/release/bootstrap.py +509 -0
- techtree/release/checks.py +376 -0
- techtree/release/document.py +125 -0
- techtree/release/generate.py +221 -0
- techtree/release/models.py +293 -0
- techtree/release/provenance.py +109 -0
- techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
- techtree/resources/catalog/catalog.json +32 -0
- techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
- techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
- techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
- techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
- techtree/resources/engines/default/engine.json +20 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
- techtree/resources/engines/default/pyproject.toml +23 -0
- techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
- techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
- techtree/resources/engines/default/tools/normalize_validation.py +222 -0
- techtree/resources/engines/default/uv.lock +1758 -0
- techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
- techtree/resources/release/build-provenance.json +4 -0
- techtree/resources/release/release-core.json +24 -0
- techtree/runs/__init__.py +31 -0
- techtree/runs/artifacts.py +750 -0
- techtree/runs/child_registry.py +228 -0
- techtree/runs/events.py +478 -0
- techtree/runs/executor.py +140 -0
- techtree/runs/fake.py +741 -0
- techtree/runs/launcher.py +253 -0
- techtree/runs/machine.py +489 -0
- techtree/runs/real.py +789 -0
- techtree/runs/service.py +616 -0
- techtree/runs/store.py +555 -0
- techtree/runs/validation.py +259 -0
- techtree/runs/variants.py +684 -0
- techtree/settings.py +143 -0
- techtree/skills/__init__.py +14 -0
- techtree/skills/archive.py +282 -0
- techtree/skills/policy.py +62 -0
- techtree/skills/scanner.py +394 -0
- techtree/skills/service.py +752 -0
- techtree/skills/starter.py +434 -0
- techtree/tasksets/__init__.py +1 -0
- techtree/tasksets/membership.py +269 -0
- techtree/tasksets/provider.py +207 -0
- techtree/tasksets/resolver.py +311 -0
- techtree/tasksets/service.py +484 -0
- techtree/tasksets/verifiers_cli.py +538 -0
- techtree/uplift/__init__.py +20 -0
- techtree/uplift/context.py +544 -0
- techtree/uplift/derive.py +203 -0
- techtree/uplift/public_tasks.py +151 -0
- techtree/uplift/service.py +719 -0
- techtree/uplift/source.py +160 -0
- techtree/verifiers/__init__.py +31 -0
- techtree/verifiers/budget.py +219 -0
- techtree/verifiers/child.py +633 -0
- techtree/verifiers/compiler.py +432 -0
- techtree/verifiers/config.py +365 -0
- techtree/verifiers/credentials.py +321 -0
- techtree/verifiers/image.py +126 -0
- techtree/verifiers/models.py +527 -0
- techtree/verifiers/outputs.py +368 -0
- techtree/verifiers/progress.py +192 -0
- techtree/verifiers/supervisor.py +341 -0
- techtree/verifiers/verify.py +782 -0
- techtree/version.py +39 -0
- techtree/worker/__init__.py +18 -0
- techtree/worker/execute.py +487 -0
- techtree/worker/main.py +57 -0
- techtree-0.1.0.dist-info/METADATA +344 -0
- techtree-0.1.0.dist-info/RECORD +174 -0
- techtree-0.1.0.dist-info/WHEEL +4 -0
- techtree-0.1.0.dist-info/entry_points.txt +3 -0
- techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,630 @@
|
|
|
1
|
+
"""What one comparison cost to run. Decisions document 0007 R6+R8.
|
|
2
|
+
|
|
3
|
+
A run's receipts say what was scored. They say nothing about what the scoring
|
|
4
|
+
consumed — how long each side took, how far apart the two children started, how
|
|
5
|
+
many tokens went out, what the provider charged — and a participant deciding
|
|
6
|
+
whether to run a second comparison needs exactly that. Decisions document 0007
|
|
7
|
+
R6 answers it with a new artifact rather than by reopening the receipts: one
|
|
8
|
+
signed, content-addressed ``techtree.comparison-execution.v1alpha1`` record per
|
|
9
|
+
comparison, written beside the report and carried in the proof bundle.
|
|
10
|
+
|
|
11
|
+
Four rules decide everything in this module.
|
|
12
|
+
|
|
13
|
+
*It is operational evidence, and it is orthogonal to reward truth.* Nothing
|
|
14
|
+
here is an input to a score, a decision, or a comparison status. A record that
|
|
15
|
+
is missing, incomplete, or absent entirely leaves the measurement exactly as it
|
|
16
|
+
was: R6 is explicit that missing economics makes cost and timing unavailable
|
|
17
|
+
with an operational-evidence warning, and never invalidates a valid score. That
|
|
18
|
+
is why no function here can fail a run.
|
|
19
|
+
|
|
20
|
+
*Provenance is stated, never implied.* Every cost carries one of four
|
|
21
|
+
provenances — provider-reported, computed from a pinned price, estimated, or
|
|
22
|
+
unavailable — and an estimate is never shown as provider-reported. When two
|
|
23
|
+
variants' costs are added together the sum takes the *weakest* provenance of
|
|
24
|
+
the two, because a total that mixed a reported number with a guess and called
|
|
25
|
+
itself reported would be the exact misstatement R6 forbids.
|
|
26
|
+
|
|
27
|
+
*Usage is what the evidence carries, and the coverage says how much.* Token
|
|
28
|
+
counts are summed from the engine's own normalized per-trace usage. Traces that
|
|
29
|
+
report none are counted rather than treated as zero, so a partial record is
|
|
30
|
+
visible as partial. Model calls are counted separately, because every trace
|
|
31
|
+
records them and tokens are not always reported alongside: a variant can
|
|
32
|
+
honestly know how many calls it made and not know what they consumed.
|
|
33
|
+
|
|
34
|
+
*Nothing is derived from a wall clock twice.* Elapsed times come from the
|
|
35
|
+
child outcomes the run already recorded, launch skew from the operational
|
|
36
|
+
record the scheduler already wrote, and the overlap from the two intervals. No
|
|
37
|
+
timestamp is taken here, so building the record twice from one run produces
|
|
38
|
+
identical bytes.
|
|
39
|
+
|
|
40
|
+
The record is not part of the frozen v0.1 protocol, and lives here for the same
|
|
41
|
+
reason :class:`~techtree.receipts.bundle.LocalProofBundleManifest` does: the
|
|
42
|
+
protocol has no such object, and inventing one inside ``models/`` would be an
|
|
43
|
+
unratified amendment. R6 leaves the link from ``UpliftReport`` to a later
|
|
44
|
+
protocol revision, so nothing in the report points at this record; the bundle's
|
|
45
|
+
signed manifest is what binds it to the run.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
from __future__ import annotations
|
|
49
|
+
|
|
50
|
+
import json
|
|
51
|
+
from collections.abc import Mapping, Sequence
|
|
52
|
+
from datetime import datetime
|
|
53
|
+
from enum import StrEnum
|
|
54
|
+
from pathlib import Path
|
|
55
|
+
from typing import Final, Literal, Self
|
|
56
|
+
|
|
57
|
+
from pydantic import Field, model_validator
|
|
58
|
+
from pydantic import ValidationError as PydanticValidationError
|
|
59
|
+
|
|
60
|
+
from techtree.models.base import (
|
|
61
|
+
Digest,
|
|
62
|
+
NonEmptyString,
|
|
63
|
+
ObjectEnvelope,
|
|
64
|
+
ProtocolModel,
|
|
65
|
+
UtcDateTime,
|
|
66
|
+
)
|
|
67
|
+
from techtree.models.campaign import VariantSchedule
|
|
68
|
+
from techtree.models.experiment import ExperimentVariant
|
|
69
|
+
from techtree.runs.child_registry import children_record_path
|
|
70
|
+
from techtree.verifiers.compiler import divide_concurrency
|
|
71
|
+
from techtree.verifiers.models import (
|
|
72
|
+
RealExecutionResult,
|
|
73
|
+
VariantExecutionResult,
|
|
74
|
+
VariantName,
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
__all__ = [
|
|
78
|
+
"COMPARISON_EXECUTION_RECORD_INVALID",
|
|
79
|
+
"COMPARISON_EXECUTION_SCHEMA_VERSION",
|
|
80
|
+
"EXECUTION_RECORD_FILENAME",
|
|
81
|
+
"OPERATIONAL_EVIDENCE_UNAVAILABLE",
|
|
82
|
+
"ComparisonExecutionRecord",
|
|
83
|
+
"CostProvenance",
|
|
84
|
+
"PairOutcome",
|
|
85
|
+
"TotalCost",
|
|
86
|
+
"UsageProvenance",
|
|
87
|
+
"VariantCost",
|
|
88
|
+
"VariantExecutionSummary",
|
|
89
|
+
"VariantUsage",
|
|
90
|
+
"build_comparison_execution_record",
|
|
91
|
+
"read_children_record",
|
|
92
|
+
"read_execution_record",
|
|
93
|
+
"unavailable_cost",
|
|
94
|
+
"weakest_provenance",
|
|
95
|
+
]
|
|
96
|
+
|
|
97
|
+
#: This record's own schema version. Not in :mod:`techtree.constants`, which
|
|
98
|
+
#: holds protocol schema versions, and this is not a protocol object.
|
|
99
|
+
COMPARISON_EXECUTION_SCHEMA_VERSION: Final = "techtree.comparison-execution.v1alpha1"
|
|
100
|
+
|
|
101
|
+
#: Stable error code for a record that does not hold together. Used by the
|
|
102
|
+
#: bundle verifier; nothing raises it while a run is being scored, because a
|
|
103
|
+
#: broken operational record never invalidates a measurement.
|
|
104
|
+
COMPARISON_EXECUTION_RECORD_INVALID: Final = "comparison_execution_record_invalid"
|
|
105
|
+
|
|
106
|
+
#: What a reader is told when a bundle carries no operational record at all.
|
|
107
|
+
#: Decisions document 0007 R6: this is a warning about what is unknown, never
|
|
108
|
+
#: a finding about what was measured.
|
|
109
|
+
OPERATIONAL_EVIDENCE_UNAVAILABLE: Final = "operational_evidence_unavailable"
|
|
110
|
+
|
|
111
|
+
#: Where the record is placed inside a proof bundle. Owned here rather than by
|
|
112
|
+
#: the bundle module, because the record's own reader needs it and a module
|
|
113
|
+
#: that names a file should be the one that writes it.
|
|
114
|
+
EXECUTION_RECORD_FILENAME: Final = "comparison-execution.json"
|
|
115
|
+
|
|
116
|
+
#: Both sides, always in comparison order.
|
|
117
|
+
_VARIANT_ORDER: Final[tuple[VariantName, ...]] = (
|
|
118
|
+
VariantName.BASELINE,
|
|
119
|
+
VariantName.CANDIDATE,
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
class CostProvenance(StrEnum):
|
|
124
|
+
"""Where a cost figure came from. Decisions document 0007 R6.
|
|
125
|
+
|
|
126
|
+
The four values are ordered by how much they claim, and the order is
|
|
127
|
+
load-bearing: :func:`weakest_provenance` uses it so that a total can never
|
|
128
|
+
describe itself as better sourced than the worst number inside it.
|
|
129
|
+
"""
|
|
130
|
+
|
|
131
|
+
#: The provider billed this and said so.
|
|
132
|
+
PROVIDER_REPORTED = "provider_reported"
|
|
133
|
+
#: Computed from recorded usage and a price the release pinned.
|
|
134
|
+
COMPUTED_FROM_PINNED_PRICE = "computed_from_pinned_price"
|
|
135
|
+
#: A figure Techtree worked out for guidance. Never presented as billed.
|
|
136
|
+
ESTIMATED = "estimated"
|
|
137
|
+
#: No cost is known. The comparison is unaffected.
|
|
138
|
+
UNAVAILABLE = "unavailable"
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
#: Weakest first, so ``min`` over this order answers "what may the sum claim?".
|
|
142
|
+
_PROVENANCE_STRENGTH: Final[dict[CostProvenance, int]] = {
|
|
143
|
+
CostProvenance.UNAVAILABLE: 0,
|
|
144
|
+
CostProvenance.ESTIMATED: 1,
|
|
145
|
+
CostProvenance.COMPUTED_FROM_PINNED_PRICE: 2,
|
|
146
|
+
CostProvenance.PROVIDER_REPORTED: 3,
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
class UsageProvenance(StrEnum):
|
|
151
|
+
"""Where token counts came from."""
|
|
152
|
+
|
|
153
|
+
#: Summed from the engine's normalized per-trace usage.
|
|
154
|
+
NORMALIZED_TRACES = "normalized_traces"
|
|
155
|
+
#: No trace in this variant reported any usage.
|
|
156
|
+
UNAVAILABLE = "unavailable"
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
class PairOutcome(StrEnum):
|
|
160
|
+
"""How the pair of children ended."""
|
|
161
|
+
|
|
162
|
+
COMPLETED = "completed"
|
|
163
|
+
CANCELLED = "cancelled"
|
|
164
|
+
FAILED = "failed"
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
class VariantCost(ProtocolModel):
|
|
168
|
+
"""One side's cost, and the honest account of where the number is from."""
|
|
169
|
+
|
|
170
|
+
cost_usd: float | None = Field(default=None, ge=0.0)
|
|
171
|
+
provenance: CostProvenance
|
|
172
|
+
detail: NonEmptyString
|
|
173
|
+
|
|
174
|
+
@model_validator(mode="after")
|
|
175
|
+
def _check_the_number_and_its_provenance_agree(self) -> Self:
|
|
176
|
+
"""Reject a cost that claims a source it has no figure for, or vice versa."""
|
|
177
|
+
known = self.cost_usd is not None
|
|
178
|
+
claims_source = self.provenance is not CostProvenance.UNAVAILABLE
|
|
179
|
+
if known != claims_source:
|
|
180
|
+
raise ValueError(
|
|
181
|
+
"a cost figure needs a provenance and a provenance needs a "
|
|
182
|
+
f"figure; got {self.cost_usd!r} as {self.provenance.value}"
|
|
183
|
+
)
|
|
184
|
+
return self
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
class TotalCost(ProtocolModel):
|
|
188
|
+
"""Both sides' cost added up, at the weakest provenance of the two."""
|
|
189
|
+
|
|
190
|
+
cost_usd: float | None = Field(default=None, ge=0.0)
|
|
191
|
+
provenance: CostProvenance
|
|
192
|
+
|
|
193
|
+
@model_validator(mode="after")
|
|
194
|
+
def _check_the_number_and_its_provenance_agree(self) -> Self:
|
|
195
|
+
"""Reject a total that claims a source it has no figure for."""
|
|
196
|
+
if (self.cost_usd is not None) != (
|
|
197
|
+
self.provenance is not CostProvenance.UNAVAILABLE
|
|
198
|
+
):
|
|
199
|
+
raise ValueError("a total cost figure needs a provenance, and vice versa")
|
|
200
|
+
return self
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
class VariantUsage(ProtocolModel):
|
|
204
|
+
"""What one side consumed, as the normalized evidence recorded it.
|
|
205
|
+
|
|
206
|
+
``traces_total`` and ``traces_with_usage`` are carried so that partial
|
|
207
|
+
coverage is legible: a variant where three traces of thirty-six reported
|
|
208
|
+
tokens is not a variant whose token count means what a complete one means,
|
|
209
|
+
and a reader is told that rather than left to assume it.
|
|
210
|
+
"""
|
|
211
|
+
|
|
212
|
+
provenance: UsageProvenance
|
|
213
|
+
model_calls: int | None = Field(default=None, ge=0)
|
|
214
|
+
input_tokens: int | None = Field(default=None, ge=0)
|
|
215
|
+
cached_input_tokens: int | None = Field(default=None, ge=0)
|
|
216
|
+
output_tokens: int | None = Field(default=None, ge=0)
|
|
217
|
+
total_tokens: int | None = Field(default=None, ge=0)
|
|
218
|
+
traces_total: int = Field(ge=0)
|
|
219
|
+
traces_with_usage: int = Field(ge=0)
|
|
220
|
+
|
|
221
|
+
@model_validator(mode="after")
|
|
222
|
+
def _check_the_counts_and_the_provenance_agree(self) -> Self:
|
|
223
|
+
"""Reject a usage block that reports numbers it says it does not have."""
|
|
224
|
+
if self.traces_with_usage > self.traces_total:
|
|
225
|
+
raise ValueError(
|
|
226
|
+
"a variant cannot have more traces reporting usage than traces"
|
|
227
|
+
)
|
|
228
|
+
reported = self.provenance is UsageProvenance.NORMALIZED_TRACES
|
|
229
|
+
if reported != (self.total_tokens is not None):
|
|
230
|
+
raise ValueError(
|
|
231
|
+
"token totals are present exactly when usage was reported; got "
|
|
232
|
+
f"{self.total_tokens!r} as {self.provenance.value}"
|
|
233
|
+
)
|
|
234
|
+
if reported and self.traces_with_usage == 0:
|
|
235
|
+
raise ValueError(
|
|
236
|
+
"usage cannot come from normalized traces when no trace reported any"
|
|
237
|
+
)
|
|
238
|
+
return self
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
class VariantExecutionSummary(ProtocolModel):
|
|
242
|
+
"""Everything one side of the comparison did, operationally."""
|
|
243
|
+
|
|
244
|
+
variant: ExperimentVariant
|
|
245
|
+
started_at: UtcDateTime
|
|
246
|
+
finished_at: UtcDateTime
|
|
247
|
+
elapsed_seconds: float = Field(ge=0.0)
|
|
248
|
+
exit_code: int
|
|
249
|
+
cancelled: bool
|
|
250
|
+
episode_count: int = Field(ge=0)
|
|
251
|
+
max_concurrent: int = Field(ge=1)
|
|
252
|
+
usage: VariantUsage
|
|
253
|
+
cost: VariantCost
|
|
254
|
+
experiment_manifest_digest: Digest
|
|
255
|
+
argv_digest: Digest
|
|
256
|
+
normalized_episodes_digest: Digest
|
|
257
|
+
raw_traces_digest: Digest
|
|
258
|
+
resolved_config_digest: Digest
|
|
259
|
+
|
|
260
|
+
@model_validator(mode="after")
|
|
261
|
+
def _check_the_clock_moves_forward(self) -> Self:
|
|
262
|
+
"""Reject a side that finished before it started."""
|
|
263
|
+
if self.finished_at < self.started_at:
|
|
264
|
+
raise ValueError("a variant cannot finish before it starts")
|
|
265
|
+
return self
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
class ComparisonExecutionRecord(ProtocolModel):
|
|
269
|
+
"""One comparison's operational record. Decisions document 0007 R6+R8."""
|
|
270
|
+
|
|
271
|
+
schema_version: Literal["techtree.comparison-execution.v1alpha1"]
|
|
272
|
+
run_id: NonEmptyString
|
|
273
|
+
campaign_spec_digest: Digest
|
|
274
|
+
engine_digest: Digest
|
|
275
|
+
execution_backend: Literal["verifiers"]
|
|
276
|
+
schedule: VariantSchedule
|
|
277
|
+
started_at: UtcDateTime
|
|
278
|
+
finished_at: UtcDateTime
|
|
279
|
+
elapsed_seconds: float = Field(ge=0.0)
|
|
280
|
+
launch_skew_seconds: float | None = Field(default=None, ge=0.0)
|
|
281
|
+
first_launched: ExperimentVariant | None
|
|
282
|
+
overlap_seconds: float = Field(ge=0.0)
|
|
283
|
+
campaign_max_concurrent: int = Field(ge=1)
|
|
284
|
+
outcome: PairOutcome
|
|
285
|
+
baseline: VariantExecutionSummary
|
|
286
|
+
candidate: VariantExecutionSummary
|
|
287
|
+
|
|
288
|
+
@model_validator(mode="after")
|
|
289
|
+
def _check_each_side_is_the_side_it_claims(self) -> Self:
|
|
290
|
+
"""Reject a record that files a variant under the wrong name."""
|
|
291
|
+
if self.baseline.variant is not ExperimentVariant.BASELINE:
|
|
292
|
+
raise ValueError("the baseline slot holds the baseline variant")
|
|
293
|
+
if self.candidate.variant is not ExperimentVariant.CANDIDATE:
|
|
294
|
+
raise ValueError("the candidate slot holds the candidate variant")
|
|
295
|
+
if (self.launch_skew_seconds is None) != (self.first_launched is None):
|
|
296
|
+
raise ValueError(
|
|
297
|
+
"a launch skew names which side went first, and a side that "
|
|
298
|
+
"went first implies a skew"
|
|
299
|
+
)
|
|
300
|
+
return self
|
|
301
|
+
|
|
302
|
+
@property
|
|
303
|
+
def total_cost(self) -> TotalCost:
|
|
304
|
+
"""Return both sides added up, claiming only what the weaker side can.
|
|
305
|
+
|
|
306
|
+
A total that mixed a provider's own figure with an estimate and called
|
|
307
|
+
itself provider-reported would be the one misstatement decisions
|
|
308
|
+
document 0007 R6 names outright, so the sum takes the weakest
|
|
309
|
+
provenance of the two and is absent entirely when either side is.
|
|
310
|
+
"""
|
|
311
|
+
provenance = weakest_provenance(
|
|
312
|
+
[self.baseline.cost.provenance, self.candidate.cost.provenance]
|
|
313
|
+
)
|
|
314
|
+
if provenance is CostProvenance.UNAVAILABLE:
|
|
315
|
+
return TotalCost(cost_usd=None, provenance=provenance)
|
|
316
|
+
# Both sides carry a figure: the validator on VariantCost guarantees a
|
|
317
|
+
# non-unavailable provenance has one.
|
|
318
|
+
assert self.baseline.cost.cost_usd is not None
|
|
319
|
+
assert self.candidate.cost.cost_usd is not None
|
|
320
|
+
return TotalCost(
|
|
321
|
+
cost_usd=self.baseline.cost.cost_usd + self.candidate.cost.cost_usd,
|
|
322
|
+
provenance=provenance,
|
|
323
|
+
)
|
|
324
|
+
|
|
325
|
+
@property
|
|
326
|
+
def total_tokens(self) -> int | None:
|
|
327
|
+
"""Return both sides' tokens, or ``None`` when either side has none."""
|
|
328
|
+
if self.baseline.usage.total_tokens is None:
|
|
329
|
+
return None
|
|
330
|
+
if self.candidate.usage.total_tokens is None:
|
|
331
|
+
return None
|
|
332
|
+
return self.baseline.usage.total_tokens + self.candidate.usage.total_tokens
|
|
333
|
+
|
|
334
|
+
def side(self, variant: ExperimentVariant) -> VariantExecutionSummary:
|
|
335
|
+
"""Return one side's summary."""
|
|
336
|
+
return (
|
|
337
|
+
self.baseline if variant is ExperimentVariant.BASELINE else self.candidate
|
|
338
|
+
)
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def weakest_provenance(provenances: Sequence[CostProvenance]) -> CostProvenance:
|
|
342
|
+
"""Return the least-claiming provenance among several."""
|
|
343
|
+
return min(provenances, key=lambda value: _PROVENANCE_STRENGTH[value])
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def unavailable_cost(detail: str) -> VariantCost:
|
|
347
|
+
"""Return the cost of a variant whose economics nothing reported."""
|
|
348
|
+
return VariantCost(
|
|
349
|
+
cost_usd=None, provenance=CostProvenance.UNAVAILABLE, detail=detail
|
|
350
|
+
)
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
#: What this build says when it has no economics at all. Stated once so the
|
|
354
|
+
#: sentence a reader meets is the same everywhere, and so the day a price feed
|
|
355
|
+
#: lands there is one place that stops being true.
|
|
356
|
+
NO_COST_SOURCE: Final = (
|
|
357
|
+
"no cost figure was reported for this variant, and this build pins no "
|
|
358
|
+
"price to compute one from"
|
|
359
|
+
)
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
# ---------------------------------------------------------------------------
|
|
363
|
+
# Building
|
|
364
|
+
# ---------------------------------------------------------------------------
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def build_comparison_execution_record(
|
|
368
|
+
*,
|
|
369
|
+
run_id: str,
|
|
370
|
+
campaign_spec_digest: Digest,
|
|
371
|
+
campaign_max_concurrent: int,
|
|
372
|
+
execution: RealExecutionResult,
|
|
373
|
+
run_root: Path,
|
|
374
|
+
costs: Mapping[VariantName, VariantCost] | None = None,
|
|
375
|
+
) -> ComparisonExecutionRecord:
|
|
376
|
+
"""Assemble one comparison's operational record from what the run recorded.
|
|
377
|
+
|
|
378
|
+
Every value is read from something the run already wrote: the child
|
|
379
|
+
outcomes, the normalized episodes, the artifact references beside them,
|
|
380
|
+
and the scheduler's own children record. ``costs`` is the seam a price
|
|
381
|
+
feed arrives through; with nothing passed, both sides report an
|
|
382
|
+
unavailable cost, which is what this build's runs honestly produce.
|
|
383
|
+
"""
|
|
384
|
+
skew_seconds, first_launched = read_children_record(run_root)
|
|
385
|
+
sides = {
|
|
386
|
+
variant: _summary(
|
|
387
|
+
result=_side(execution, variant),
|
|
388
|
+
campaign_max_concurrent=campaign_max_concurrent,
|
|
389
|
+
schedule=execution.schedule,
|
|
390
|
+
variant=variant,
|
|
391
|
+
cost=(costs or {}).get(variant),
|
|
392
|
+
)
|
|
393
|
+
for variant in _VARIANT_ORDER
|
|
394
|
+
}
|
|
395
|
+
baseline = sides[VariantName.BASELINE]
|
|
396
|
+
candidate = sides[VariantName.CANDIDATE]
|
|
397
|
+
|
|
398
|
+
started_at = min(baseline.started_at, candidate.started_at)
|
|
399
|
+
finished_at = max(baseline.finished_at, candidate.finished_at)
|
|
400
|
+
return ComparisonExecutionRecord(
|
|
401
|
+
schema_version=COMPARISON_EXECUTION_SCHEMA_VERSION,
|
|
402
|
+
run_id=run_id,
|
|
403
|
+
campaign_spec_digest=campaign_spec_digest,
|
|
404
|
+
engine_digest=execution.engine_digest,
|
|
405
|
+
execution_backend=execution.execution_backend,
|
|
406
|
+
schedule=execution.schedule,
|
|
407
|
+
started_at=started_at,
|
|
408
|
+
finished_at=finished_at,
|
|
409
|
+
elapsed_seconds=_seconds(started_at, finished_at),
|
|
410
|
+
launch_skew_seconds=skew_seconds,
|
|
411
|
+
first_launched=first_launched,
|
|
412
|
+
overlap_seconds=_overlap(baseline, candidate),
|
|
413
|
+
campaign_max_concurrent=campaign_max_concurrent,
|
|
414
|
+
outcome=_outcome(baseline, candidate),
|
|
415
|
+
baseline=baseline,
|
|
416
|
+
candidate=candidate,
|
|
417
|
+
)
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def read_execution_record(bundle_dir: Path) -> ComparisonExecutionRecord | None:
|
|
421
|
+
"""Return the record one proof bundle carries, or ``None`` when it has none.
|
|
422
|
+
|
|
423
|
+
Reading is deliberately separate from checking. The bundle verifier is
|
|
424
|
+
what decides whether a record is signed, intact and about this run; this
|
|
425
|
+
is what a renderer calls afterwards, and it returns nothing rather than
|
|
426
|
+
raising, because a result whose economics cannot be read is still a
|
|
427
|
+
result. A bundle whose record does not verify is already reported by the
|
|
428
|
+
verification the caller ran before it got here.
|
|
429
|
+
"""
|
|
430
|
+
path = bundle_dir / EXECUTION_RECORD_FILENAME
|
|
431
|
+
try:
|
|
432
|
+
raw = path.read_bytes()
|
|
433
|
+
except OSError:
|
|
434
|
+
return None
|
|
435
|
+
try:
|
|
436
|
+
envelope = ObjectEnvelope[ComparisonExecutionRecord].model_validate_json(raw)
|
|
437
|
+
except PydanticValidationError:
|
|
438
|
+
return None
|
|
439
|
+
return envelope.payload
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
def read_children_record(
|
|
443
|
+
run_root: Path,
|
|
444
|
+
) -> tuple[float | None, ExperimentVariant | None]:
|
|
445
|
+
"""Return the launch skew and which side went first, if the run recorded them.
|
|
446
|
+
|
|
447
|
+
The scheduler writes this while the children are being started, which is
|
|
448
|
+
the only moment either value can be observed. A run without the file — a
|
|
449
|
+
sequential schedule records no skew, and an older run may have written
|
|
450
|
+
none — reports neither rather than reconstructing one from timestamps that
|
|
451
|
+
were taken for something else.
|
|
452
|
+
"""
|
|
453
|
+
path = children_record_path(run_root)
|
|
454
|
+
try:
|
|
455
|
+
document = json.loads(path.read_bytes())
|
|
456
|
+
except (OSError, ValueError):
|
|
457
|
+
return None, None
|
|
458
|
+
if not isinstance(document, dict):
|
|
459
|
+
return None, None
|
|
460
|
+
|
|
461
|
+
skew = document.get("launch_skew_seconds")
|
|
462
|
+
if not isinstance(skew, int | float) or isinstance(skew, bool) or skew < 0:
|
|
463
|
+
return None, None
|
|
464
|
+
|
|
465
|
+
started = [
|
|
466
|
+
(row.get("started_at"), row.get("variant"))
|
|
467
|
+
for row in document.get("children", [])
|
|
468
|
+
if isinstance(row, dict)
|
|
469
|
+
]
|
|
470
|
+
first = _first_launched(started)
|
|
471
|
+
if first is None:
|
|
472
|
+
return None, None
|
|
473
|
+
return float(skew), first
|
|
474
|
+
|
|
475
|
+
|
|
476
|
+
# ---------------------------------------------------------------------------
|
|
477
|
+
# The pieces
|
|
478
|
+
# ---------------------------------------------------------------------------
|
|
479
|
+
|
|
480
|
+
|
|
481
|
+
def _summary(
|
|
482
|
+
*,
|
|
483
|
+
result: VariantExecutionResult,
|
|
484
|
+
campaign_max_concurrent: int,
|
|
485
|
+
schedule: VariantSchedule,
|
|
486
|
+
variant: VariantName,
|
|
487
|
+
cost: VariantCost | None,
|
|
488
|
+
) -> VariantExecutionSummary:
|
|
489
|
+
"""Describe one side from its own recorded evidence."""
|
|
490
|
+
outcome = result.child_outcome
|
|
491
|
+
baseline_permits, candidate_permits = divide_concurrency(
|
|
492
|
+
schedule, campaign_max_concurrent
|
|
493
|
+
)
|
|
494
|
+
return VariantExecutionSummary(
|
|
495
|
+
variant=_protocol_variant(variant),
|
|
496
|
+
started_at=outcome.started_at,
|
|
497
|
+
finished_at=outcome.finished_at,
|
|
498
|
+
elapsed_seconds=_seconds(outcome.started_at, outcome.finished_at),
|
|
499
|
+
exit_code=outcome.exit_code,
|
|
500
|
+
cancelled=outcome.cancelled,
|
|
501
|
+
episode_count=len(result.episodes),
|
|
502
|
+
max_concurrent=(
|
|
503
|
+
baseline_permits if variant is VariantName.BASELINE else candidate_permits
|
|
504
|
+
),
|
|
505
|
+
usage=_usage(result),
|
|
506
|
+
cost=cost or unavailable_cost(NO_COST_SOURCE),
|
|
507
|
+
experiment_manifest_digest=result.experiment_manifest_digest,
|
|
508
|
+
argv_digest=outcome.argv_digest,
|
|
509
|
+
normalized_episodes_digest=result.normalized_episodes.digest,
|
|
510
|
+
raw_traces_digest=result.raw_traces.digest,
|
|
511
|
+
resolved_config_digest=result.resolved_verifiers_config.digest,
|
|
512
|
+
)
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
def _usage(result: VariantExecutionResult) -> VariantUsage:
|
|
516
|
+
"""Sum one side's consumption from the engine's normalized traces.
|
|
517
|
+
|
|
518
|
+
Two measurements with two different availabilities, kept apart. Model
|
|
519
|
+
calls are counted by every trace, so they are known whenever this variant
|
|
520
|
+
recorded a trace at all. Token usage is reported per trace and may be
|
|
521
|
+
absent, so a trace that reports none is *counted* rather than read as a
|
|
522
|
+
zero, and the coverage travels with the totals.
|
|
523
|
+
"""
|
|
524
|
+
traces_total = 0
|
|
525
|
+
traces_with_usage = 0
|
|
526
|
+
model_calls = 0
|
|
527
|
+
totals = {"input": 0, "cached": 0, "output": 0, "total": 0}
|
|
528
|
+
for episode in result.episodes:
|
|
529
|
+
for trace in episode.traces:
|
|
530
|
+
traces_total += 1
|
|
531
|
+
model_calls += trace.model_calls
|
|
532
|
+
usage = trace.usage
|
|
533
|
+
if usage is None:
|
|
534
|
+
continue
|
|
535
|
+
traces_with_usage += 1
|
|
536
|
+
totals["input"] += usage.input_tokens
|
|
537
|
+
totals["cached"] += usage.cached_input_tokens or 0
|
|
538
|
+
totals["output"] += usage.output_tokens
|
|
539
|
+
totals["total"] += usage.total_tokens
|
|
540
|
+
|
|
541
|
+
counted = None if traces_total == 0 else model_calls
|
|
542
|
+
if traces_with_usage == 0:
|
|
543
|
+
return VariantUsage(
|
|
544
|
+
provenance=UsageProvenance.UNAVAILABLE,
|
|
545
|
+
model_calls=counted,
|
|
546
|
+
traces_total=traces_total,
|
|
547
|
+
traces_with_usage=0,
|
|
548
|
+
)
|
|
549
|
+
return VariantUsage(
|
|
550
|
+
provenance=UsageProvenance.NORMALIZED_TRACES,
|
|
551
|
+
model_calls=counted,
|
|
552
|
+
input_tokens=totals["input"],
|
|
553
|
+
cached_input_tokens=totals["cached"],
|
|
554
|
+
output_tokens=totals["output"],
|
|
555
|
+
total_tokens=totals["total"],
|
|
556
|
+
traces_total=traces_total,
|
|
557
|
+
traces_with_usage=traces_with_usage,
|
|
558
|
+
)
|
|
559
|
+
|
|
560
|
+
|
|
561
|
+
def _overlap(
|
|
562
|
+
baseline: VariantExecutionSummary, candidate: VariantExecutionSummary
|
|
563
|
+
) -> float:
|
|
564
|
+
"""Return how long both sides were running at once.
|
|
565
|
+
|
|
566
|
+
A parallel comparison's whole claim is that the two sides met the same
|
|
567
|
+
conditions, and the overlap is the measurable part of it. A sequential
|
|
568
|
+
schedule produces zero here, which is the true answer rather than a
|
|
569
|
+
missing one.
|
|
570
|
+
"""
|
|
571
|
+
start = max(baseline.started_at, candidate.started_at)
|
|
572
|
+
finish = min(baseline.finished_at, candidate.finished_at)
|
|
573
|
+
return _seconds(start, finish) if finish > start else 0.0
|
|
574
|
+
|
|
575
|
+
|
|
576
|
+
def _outcome(
|
|
577
|
+
baseline: VariantExecutionSummary, candidate: VariantExecutionSummary
|
|
578
|
+
) -> PairOutcome:
|
|
579
|
+
"""Say how the pair ended, from what the children themselves recorded."""
|
|
580
|
+
if baseline.cancelled or candidate.cancelled:
|
|
581
|
+
return PairOutcome.CANCELLED
|
|
582
|
+
if baseline.exit_code != 0 or candidate.exit_code != 0:
|
|
583
|
+
return PairOutcome.FAILED
|
|
584
|
+
return PairOutcome.COMPLETED
|
|
585
|
+
|
|
586
|
+
|
|
587
|
+
def _first_launched(
|
|
588
|
+
started: Sequence[tuple[object, object]],
|
|
589
|
+
) -> ExperimentVariant | None:
|
|
590
|
+
"""Return which variant the operational record started first."""
|
|
591
|
+
parsed: list[tuple[datetime, ExperimentVariant]] = []
|
|
592
|
+
for stamp, variant in started:
|
|
593
|
+
if not isinstance(stamp, str) or not isinstance(variant, str):
|
|
594
|
+
return None
|
|
595
|
+
try:
|
|
596
|
+
when = datetime.fromisoformat(stamp)
|
|
597
|
+
side = ExperimentVariant(variant)
|
|
598
|
+
except ValueError:
|
|
599
|
+
return None
|
|
600
|
+
parsed.append((when, side))
|
|
601
|
+
if len(parsed) != len(_VARIANT_ORDER):
|
|
602
|
+
return None
|
|
603
|
+
return min(parsed, key=lambda item: item[0])[1]
|
|
604
|
+
|
|
605
|
+
|
|
606
|
+
def _seconds(start: datetime, finish: datetime) -> float:
|
|
607
|
+
"""Return a non-negative duration between two recorded instants."""
|
|
608
|
+
return max(0.0, (finish - start).total_seconds())
|
|
609
|
+
|
|
610
|
+
|
|
611
|
+
def _side(
|
|
612
|
+
execution: RealExecutionResult, variant: VariantName
|
|
613
|
+
) -> VariantExecutionResult:
|
|
614
|
+
"""Return one side of an execution result."""
|
|
615
|
+
return (
|
|
616
|
+
execution.baseline if variant is VariantName.BASELINE else execution.candidate
|
|
617
|
+
)
|
|
618
|
+
|
|
619
|
+
|
|
620
|
+
def _protocol_variant(variant: VariantName) -> ExperimentVariant:
|
|
621
|
+
"""Return the protocol spelling of one variant name."""
|
|
622
|
+
return (
|
|
623
|
+
ExperimentVariant.BASELINE
|
|
624
|
+
if variant is VariantName.BASELINE
|
|
625
|
+
else ExperimentVariant.CANDIDATE
|
|
626
|
+
)
|
|
627
|
+
|
|
628
|
+
|
|
629
|
+
type SignedComparisonExecutionRecord = ObjectEnvelope[ComparisonExecutionRecord]
|
|
630
|
+
"""One record inside the envelope the executor signed it with."""
|