techtree 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. techtree/__init__.py +35 -0
  2. techtree/__main__.py +14 -0
  3. techtree/canonical.py +239 -0
  4. techtree/catalog/__init__.py +25 -0
  5. techtree/catalog/repository.py +400 -0
  6. techtree/catalog/service.py +419 -0
  7. techtree/cli/__init__.py +1 -0
  8. techtree/cli/app.py +416 -0
  9. techtree/cli/commands/__init__.py +1 -0
  10. techtree/cli/commands/climb.py +1223 -0
  11. techtree/cli/commands/doctor.py +147 -0
  12. techtree/cli/commands/engine.py +207 -0
  13. techtree/cli/commands/proof.py +556 -0
  14. techtree/cli/commands/publish.py +447 -0
  15. techtree/cli/commands/release.py +303 -0
  16. techtree/cli/commands/run.py +1067 -0
  17. techtree/cli/commands/setup.py +181 -0
  18. techtree/cli/commands/skill.py +221 -0
  19. techtree/cli/commands/uplift.py +698 -0
  20. techtree/cli/commands/withdraw.py +212 -0
  21. techtree/cli/confirm.py +47 -0
  22. techtree/cli/context.py +96 -0
  23. techtree/cli/invoke.py +220 -0
  24. techtree/cli/output.py +280 -0
  25. techtree/constants.py +138 -0
  26. techtree/crypto.py +128 -0
  27. techtree/doctor/__init__.py +1 -0
  28. techtree/doctor/checks.py +675 -0
  29. techtree/doctor/execution_checks.py +435 -0
  30. techtree/doctor/service.py +326 -0
  31. techtree/drafts/__init__.py +32 -0
  32. techtree/drafts/source.py +146 -0
  33. techtree/drafts/store.py +992 -0
  34. techtree/engines/__init__.py +1 -0
  35. techtree/engines/bundle.py +251 -0
  36. techtree/engines/installer.py +679 -0
  37. techtree/engines/registry.py +235 -0
  38. techtree/engines/runner.py +170 -0
  39. techtree/errors.py +262 -0
  40. techtree/fs.py +234 -0
  41. techtree/harness.py +108 -0
  42. techtree/identity/__init__.py +41 -0
  43. techtree/identity/models.py +113 -0
  44. techtree/identity/service.py +199 -0
  45. techtree/identity/store.py +263 -0
  46. techtree/ids.py +85 -0
  47. techtree/manifests/__init__.py +39 -0
  48. techtree/manifests/builder.py +433 -0
  49. techtree/manifests/compare.py +376 -0
  50. techtree/models/__init__.py +282 -0
  51. techtree/models/base.py +201 -0
  52. techtree/models/campaign.py +484 -0
  53. techtree/models/catalog.py +227 -0
  54. techtree/models/cli.py +151 -0
  55. techtree/models/climb.py +254 -0
  56. techtree/models/data_policy.py +130 -0
  57. techtree/models/engine.py +156 -0
  58. techtree/models/episode_receipt.py +130 -0
  59. techtree/models/evaluation_backend.py +113 -0
  60. techtree/models/experiment.py +154 -0
  61. techtree/models/run.py +214 -0
  62. techtree/models/skill.py +156 -0
  63. techtree/models/uplift_report.py +158 -0
  64. techtree/models/validation.py +299 -0
  65. techtree/paths.py +116 -0
  66. techtree/presentation/__init__.py +31 -0
  67. techtree/presentation/build.py +1242 -0
  68. techtree/presentation/compact.py +246 -0
  69. techtree/presentation/evidence.py +169 -0
  70. techtree/presentation/models.py +358 -0
  71. techtree/presentation/rich.py +312 -0
  72. techtree/presentation/sanitize.py +156 -0
  73. techtree/publication/__init__.py +44 -0
  74. techtree/publication/address.py +180 -0
  75. techtree/publication/coordinates.py +26 -0
  76. techtree/publication/journal.py +212 -0
  77. techtree/publication/keccak.py +183 -0
  78. techtree/publication/models.py +209 -0
  79. techtree/publication/offer.py +35 -0
  80. techtree/publication/service.py +618 -0
  81. techtree/publication/transport.py +296 -0
  82. techtree/publication/verify.py +242 -0
  83. techtree/publication/withdraw.py +156 -0
  84. techtree/py.typed +0 -0
  85. techtree/receipts/__init__.py +52 -0
  86. techtree/receipts/bundle.py +578 -0
  87. techtree/receipts/compare.py +1065 -0
  88. techtree/receipts/episode.py +672 -0
  89. techtree/receipts/execution.py +630 -0
  90. techtree/receipts/observed.py +474 -0
  91. techtree/receipts/set.py +336 -0
  92. techtree/receipts/uplift.py +655 -0
  93. techtree/receipts/verify.py +1055 -0
  94. techtree/release/__init__.py +9 -0
  95. techtree/release/bootstrap.py +509 -0
  96. techtree/release/checks.py +376 -0
  97. techtree/release/document.py +125 -0
  98. techtree/release/generate.py +221 -0
  99. techtree/release/models.py +293 -0
  100. techtree/release/provenance.py +109 -0
  101. techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
  102. techtree/resources/catalog/catalog.json +32 -0
  103. techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
  104. techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
  105. techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
  106. techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
  107. techtree/resources/engines/default/engine.json +20 -0
  108. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
  109. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
  110. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
  111. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
  112. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
  113. techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
  114. techtree/resources/engines/default/pyproject.toml +23 -0
  115. techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
  116. techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
  117. techtree/resources/engines/default/tools/normalize_validation.py +222 -0
  118. techtree/resources/engines/default/uv.lock +1758 -0
  119. techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
  120. techtree/resources/release/build-provenance.json +4 -0
  121. techtree/resources/release/release-core.json +24 -0
  122. techtree/runs/__init__.py +31 -0
  123. techtree/runs/artifacts.py +750 -0
  124. techtree/runs/child_registry.py +228 -0
  125. techtree/runs/events.py +478 -0
  126. techtree/runs/executor.py +140 -0
  127. techtree/runs/fake.py +741 -0
  128. techtree/runs/launcher.py +253 -0
  129. techtree/runs/machine.py +489 -0
  130. techtree/runs/real.py +789 -0
  131. techtree/runs/service.py +616 -0
  132. techtree/runs/store.py +555 -0
  133. techtree/runs/validation.py +259 -0
  134. techtree/runs/variants.py +684 -0
  135. techtree/settings.py +143 -0
  136. techtree/skills/__init__.py +14 -0
  137. techtree/skills/archive.py +282 -0
  138. techtree/skills/policy.py +62 -0
  139. techtree/skills/scanner.py +394 -0
  140. techtree/skills/service.py +752 -0
  141. techtree/skills/starter.py +434 -0
  142. techtree/tasksets/__init__.py +1 -0
  143. techtree/tasksets/membership.py +269 -0
  144. techtree/tasksets/provider.py +207 -0
  145. techtree/tasksets/resolver.py +311 -0
  146. techtree/tasksets/service.py +484 -0
  147. techtree/tasksets/verifiers_cli.py +538 -0
  148. techtree/uplift/__init__.py +20 -0
  149. techtree/uplift/context.py +544 -0
  150. techtree/uplift/derive.py +203 -0
  151. techtree/uplift/public_tasks.py +151 -0
  152. techtree/uplift/service.py +719 -0
  153. techtree/uplift/source.py +160 -0
  154. techtree/verifiers/__init__.py +31 -0
  155. techtree/verifiers/budget.py +219 -0
  156. techtree/verifiers/child.py +633 -0
  157. techtree/verifiers/compiler.py +432 -0
  158. techtree/verifiers/config.py +365 -0
  159. techtree/verifiers/credentials.py +321 -0
  160. techtree/verifiers/image.py +126 -0
  161. techtree/verifiers/models.py +527 -0
  162. techtree/verifiers/outputs.py +368 -0
  163. techtree/verifiers/progress.py +192 -0
  164. techtree/verifiers/supervisor.py +341 -0
  165. techtree/verifiers/verify.py +782 -0
  166. techtree/version.py +39 -0
  167. techtree/worker/__init__.py +18 -0
  168. techtree/worker/execute.py +487 -0
  169. techtree/worker/main.py +57 -0
  170. techtree-0.1.0.dist-info/METADATA +344 -0
  171. techtree-0.1.0.dist-info/RECORD +174 -0
  172. techtree-0.1.0.dist-info/WHEEL +4 -0
  173. techtree-0.1.0.dist-info/entry_points.txt +3 -0
  174. techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,630 @@
1
+ """What one comparison cost to run. Decisions document 0007 R6+R8.
2
+
3
+ A run's receipts say what was scored. They say nothing about what the scoring
4
+ consumed — how long each side took, how far apart the two children started, how
5
+ many tokens went out, what the provider charged — and a participant deciding
6
+ whether to run a second comparison needs exactly that. Decisions document 0007
7
+ R6 answers it with a new artifact rather than by reopening the receipts: one
8
+ signed, content-addressed ``techtree.comparison-execution.v1alpha1`` record per
9
+ comparison, written beside the report and carried in the proof bundle.
10
+
11
+ Four rules decide everything in this module.
12
+
13
+ *It is operational evidence, and it is orthogonal to reward truth.* Nothing
14
+ here is an input to a score, a decision, or a comparison status. A record that
15
+ is missing, incomplete, or absent entirely leaves the measurement exactly as it
16
+ was: R6 is explicit that missing economics makes cost and timing unavailable
17
+ with an operational-evidence warning, and never invalidates a valid score. That
18
+ is why no function here can fail a run.
19
+
20
+ *Provenance is stated, never implied.* Every cost carries one of four
21
+ provenances — provider-reported, computed from a pinned price, estimated, or
22
+ unavailable — and an estimate is never shown as provider-reported. When two
23
+ variants' costs are added together the sum takes the *weakest* provenance of
24
+ the two, because a total that mixed a reported number with a guess and called
25
+ itself reported would be the exact misstatement R6 forbids.
26
+
27
+ *Usage is what the evidence carries, and the coverage says how much.* Token
28
+ counts are summed from the engine's own normalized per-trace usage. Traces that
29
+ report none are counted rather than treated as zero, so a partial record is
30
+ visible as partial. Model calls are counted separately, because every trace
31
+ records them and tokens are not always reported alongside: a variant can
32
+ honestly know how many calls it made and not know what they consumed.
33
+
34
+ *Nothing is derived from a wall clock twice.* Elapsed times come from the
35
+ child outcomes the run already recorded, launch skew from the operational
36
+ record the scheduler already wrote, and the overlap from the two intervals. No
37
+ timestamp is taken here, so building the record twice from one run produces
38
+ identical bytes.
39
+
40
+ The record is not part of the frozen v0.1 protocol, and lives here for the same
41
+ reason :class:`~techtree.receipts.bundle.LocalProofBundleManifest` does: the
42
+ protocol has no such object, and inventing one inside ``models/`` would be an
43
+ unratified amendment. R6 leaves the link from ``UpliftReport`` to a later
44
+ protocol revision, so nothing in the report points at this record; the bundle's
45
+ signed manifest is what binds it to the run.
46
+ """
47
+
48
+ from __future__ import annotations
49
+
50
+ import json
51
+ from collections.abc import Mapping, Sequence
52
+ from datetime import datetime
53
+ from enum import StrEnum
54
+ from pathlib import Path
55
+ from typing import Final, Literal, Self
56
+
57
+ from pydantic import Field, model_validator
58
+ from pydantic import ValidationError as PydanticValidationError
59
+
60
+ from techtree.models.base import (
61
+ Digest,
62
+ NonEmptyString,
63
+ ObjectEnvelope,
64
+ ProtocolModel,
65
+ UtcDateTime,
66
+ )
67
+ from techtree.models.campaign import VariantSchedule
68
+ from techtree.models.experiment import ExperimentVariant
69
+ from techtree.runs.child_registry import children_record_path
70
+ from techtree.verifiers.compiler import divide_concurrency
71
+ from techtree.verifiers.models import (
72
+ RealExecutionResult,
73
+ VariantExecutionResult,
74
+ VariantName,
75
+ )
76
+
77
+ __all__ = [
78
+ "COMPARISON_EXECUTION_RECORD_INVALID",
79
+ "COMPARISON_EXECUTION_SCHEMA_VERSION",
80
+ "EXECUTION_RECORD_FILENAME",
81
+ "OPERATIONAL_EVIDENCE_UNAVAILABLE",
82
+ "ComparisonExecutionRecord",
83
+ "CostProvenance",
84
+ "PairOutcome",
85
+ "TotalCost",
86
+ "UsageProvenance",
87
+ "VariantCost",
88
+ "VariantExecutionSummary",
89
+ "VariantUsage",
90
+ "build_comparison_execution_record",
91
+ "read_children_record",
92
+ "read_execution_record",
93
+ "unavailable_cost",
94
+ "weakest_provenance",
95
+ ]
96
+
97
+ #: This record's own schema version. Not in :mod:`techtree.constants`, which
98
+ #: holds protocol schema versions, and this is not a protocol object.
99
+ COMPARISON_EXECUTION_SCHEMA_VERSION: Final = "techtree.comparison-execution.v1alpha1"
100
+
101
+ #: Stable error code for a record that does not hold together. Used by the
102
+ #: bundle verifier; nothing raises it while a run is being scored, because a
103
+ #: broken operational record never invalidates a measurement.
104
+ COMPARISON_EXECUTION_RECORD_INVALID: Final = "comparison_execution_record_invalid"
105
+
106
+ #: What a reader is told when a bundle carries no operational record at all.
107
+ #: Decisions document 0007 R6: this is a warning about what is unknown, never
108
+ #: a finding about what was measured.
109
+ OPERATIONAL_EVIDENCE_UNAVAILABLE: Final = "operational_evidence_unavailable"
110
+
111
+ #: Where the record is placed inside a proof bundle. Owned here rather than by
112
+ #: the bundle module, because the record's own reader needs it and a module
113
+ #: that names a file should be the one that writes it.
114
+ EXECUTION_RECORD_FILENAME: Final = "comparison-execution.json"
115
+
116
+ #: Both sides, always in comparison order.
117
+ _VARIANT_ORDER: Final[tuple[VariantName, ...]] = (
118
+ VariantName.BASELINE,
119
+ VariantName.CANDIDATE,
120
+ )
121
+
122
+
123
+ class CostProvenance(StrEnum):
124
+ """Where a cost figure came from. Decisions document 0007 R6.
125
+
126
+ The four values are ordered by how much they claim, and the order is
127
+ load-bearing: :func:`weakest_provenance` uses it so that a total can never
128
+ describe itself as better sourced than the worst number inside it.
129
+ """
130
+
131
+ #: The provider billed this and said so.
132
+ PROVIDER_REPORTED = "provider_reported"
133
+ #: Computed from recorded usage and a price the release pinned.
134
+ COMPUTED_FROM_PINNED_PRICE = "computed_from_pinned_price"
135
+ #: A figure Techtree worked out for guidance. Never presented as billed.
136
+ ESTIMATED = "estimated"
137
+ #: No cost is known. The comparison is unaffected.
138
+ UNAVAILABLE = "unavailable"
139
+
140
+
141
+ #: Weakest first, so ``min`` over this order answers "what may the sum claim?".
142
+ _PROVENANCE_STRENGTH: Final[dict[CostProvenance, int]] = {
143
+ CostProvenance.UNAVAILABLE: 0,
144
+ CostProvenance.ESTIMATED: 1,
145
+ CostProvenance.COMPUTED_FROM_PINNED_PRICE: 2,
146
+ CostProvenance.PROVIDER_REPORTED: 3,
147
+ }
148
+
149
+
150
+ class UsageProvenance(StrEnum):
151
+ """Where token counts came from."""
152
+
153
+ #: Summed from the engine's normalized per-trace usage.
154
+ NORMALIZED_TRACES = "normalized_traces"
155
+ #: No trace in this variant reported any usage.
156
+ UNAVAILABLE = "unavailable"
157
+
158
+
159
+ class PairOutcome(StrEnum):
160
+ """How the pair of children ended."""
161
+
162
+ COMPLETED = "completed"
163
+ CANCELLED = "cancelled"
164
+ FAILED = "failed"
165
+
166
+
167
+ class VariantCost(ProtocolModel):
168
+ """One side's cost, and the honest account of where the number is from."""
169
+
170
+ cost_usd: float | None = Field(default=None, ge=0.0)
171
+ provenance: CostProvenance
172
+ detail: NonEmptyString
173
+
174
+ @model_validator(mode="after")
175
+ def _check_the_number_and_its_provenance_agree(self) -> Self:
176
+ """Reject a cost that claims a source it has no figure for, or vice versa."""
177
+ known = self.cost_usd is not None
178
+ claims_source = self.provenance is not CostProvenance.UNAVAILABLE
179
+ if known != claims_source:
180
+ raise ValueError(
181
+ "a cost figure needs a provenance and a provenance needs a "
182
+ f"figure; got {self.cost_usd!r} as {self.provenance.value}"
183
+ )
184
+ return self
185
+
186
+
187
+ class TotalCost(ProtocolModel):
188
+ """Both sides' cost added up, at the weakest provenance of the two."""
189
+
190
+ cost_usd: float | None = Field(default=None, ge=0.0)
191
+ provenance: CostProvenance
192
+
193
+ @model_validator(mode="after")
194
+ def _check_the_number_and_its_provenance_agree(self) -> Self:
195
+ """Reject a total that claims a source it has no figure for."""
196
+ if (self.cost_usd is not None) != (
197
+ self.provenance is not CostProvenance.UNAVAILABLE
198
+ ):
199
+ raise ValueError("a total cost figure needs a provenance, and vice versa")
200
+ return self
201
+
202
+
203
+ class VariantUsage(ProtocolModel):
204
+ """What one side consumed, as the normalized evidence recorded it.
205
+
206
+ ``traces_total`` and ``traces_with_usage`` are carried so that partial
207
+ coverage is legible: a variant where three traces of thirty-six reported
208
+ tokens is not a variant whose token count means what a complete one means,
209
+ and a reader is told that rather than left to assume it.
210
+ """
211
+
212
+ provenance: UsageProvenance
213
+ model_calls: int | None = Field(default=None, ge=0)
214
+ input_tokens: int | None = Field(default=None, ge=0)
215
+ cached_input_tokens: int | None = Field(default=None, ge=0)
216
+ output_tokens: int | None = Field(default=None, ge=0)
217
+ total_tokens: int | None = Field(default=None, ge=0)
218
+ traces_total: int = Field(ge=0)
219
+ traces_with_usage: int = Field(ge=0)
220
+
221
+ @model_validator(mode="after")
222
+ def _check_the_counts_and_the_provenance_agree(self) -> Self:
223
+ """Reject a usage block that reports numbers it says it does not have."""
224
+ if self.traces_with_usage > self.traces_total:
225
+ raise ValueError(
226
+ "a variant cannot have more traces reporting usage than traces"
227
+ )
228
+ reported = self.provenance is UsageProvenance.NORMALIZED_TRACES
229
+ if reported != (self.total_tokens is not None):
230
+ raise ValueError(
231
+ "token totals are present exactly when usage was reported; got "
232
+ f"{self.total_tokens!r} as {self.provenance.value}"
233
+ )
234
+ if reported and self.traces_with_usage == 0:
235
+ raise ValueError(
236
+ "usage cannot come from normalized traces when no trace reported any"
237
+ )
238
+ return self
239
+
240
+
241
+ class VariantExecutionSummary(ProtocolModel):
242
+ """Everything one side of the comparison did, operationally."""
243
+
244
+ variant: ExperimentVariant
245
+ started_at: UtcDateTime
246
+ finished_at: UtcDateTime
247
+ elapsed_seconds: float = Field(ge=0.0)
248
+ exit_code: int
249
+ cancelled: bool
250
+ episode_count: int = Field(ge=0)
251
+ max_concurrent: int = Field(ge=1)
252
+ usage: VariantUsage
253
+ cost: VariantCost
254
+ experiment_manifest_digest: Digest
255
+ argv_digest: Digest
256
+ normalized_episodes_digest: Digest
257
+ raw_traces_digest: Digest
258
+ resolved_config_digest: Digest
259
+
260
+ @model_validator(mode="after")
261
+ def _check_the_clock_moves_forward(self) -> Self:
262
+ """Reject a side that finished before it started."""
263
+ if self.finished_at < self.started_at:
264
+ raise ValueError("a variant cannot finish before it starts")
265
+ return self
266
+
267
+
268
+ class ComparisonExecutionRecord(ProtocolModel):
269
+ """One comparison's operational record. Decisions document 0007 R6+R8."""
270
+
271
+ schema_version: Literal["techtree.comparison-execution.v1alpha1"]
272
+ run_id: NonEmptyString
273
+ campaign_spec_digest: Digest
274
+ engine_digest: Digest
275
+ execution_backend: Literal["verifiers"]
276
+ schedule: VariantSchedule
277
+ started_at: UtcDateTime
278
+ finished_at: UtcDateTime
279
+ elapsed_seconds: float = Field(ge=0.0)
280
+ launch_skew_seconds: float | None = Field(default=None, ge=0.0)
281
+ first_launched: ExperimentVariant | None
282
+ overlap_seconds: float = Field(ge=0.0)
283
+ campaign_max_concurrent: int = Field(ge=1)
284
+ outcome: PairOutcome
285
+ baseline: VariantExecutionSummary
286
+ candidate: VariantExecutionSummary
287
+
288
+ @model_validator(mode="after")
289
+ def _check_each_side_is_the_side_it_claims(self) -> Self:
290
+ """Reject a record that files a variant under the wrong name."""
291
+ if self.baseline.variant is not ExperimentVariant.BASELINE:
292
+ raise ValueError("the baseline slot holds the baseline variant")
293
+ if self.candidate.variant is not ExperimentVariant.CANDIDATE:
294
+ raise ValueError("the candidate slot holds the candidate variant")
295
+ if (self.launch_skew_seconds is None) != (self.first_launched is None):
296
+ raise ValueError(
297
+ "a launch skew names which side went first, and a side that "
298
+ "went first implies a skew"
299
+ )
300
+ return self
301
+
302
+ @property
303
+ def total_cost(self) -> TotalCost:
304
+ """Return both sides added up, claiming only what the weaker side can.
305
+
306
+ A total that mixed a provider's own figure with an estimate and called
307
+ itself provider-reported would be the one misstatement decisions
308
+ document 0007 R6 names outright, so the sum takes the weakest
309
+ provenance of the two and is absent entirely when either side is.
310
+ """
311
+ provenance = weakest_provenance(
312
+ [self.baseline.cost.provenance, self.candidate.cost.provenance]
313
+ )
314
+ if provenance is CostProvenance.UNAVAILABLE:
315
+ return TotalCost(cost_usd=None, provenance=provenance)
316
+ # Both sides carry a figure: the validator on VariantCost guarantees a
317
+ # non-unavailable provenance has one.
318
+ assert self.baseline.cost.cost_usd is not None
319
+ assert self.candidate.cost.cost_usd is not None
320
+ return TotalCost(
321
+ cost_usd=self.baseline.cost.cost_usd + self.candidate.cost.cost_usd,
322
+ provenance=provenance,
323
+ )
324
+
325
+ @property
326
+ def total_tokens(self) -> int | None:
327
+ """Return both sides' tokens, or ``None`` when either side has none."""
328
+ if self.baseline.usage.total_tokens is None:
329
+ return None
330
+ if self.candidate.usage.total_tokens is None:
331
+ return None
332
+ return self.baseline.usage.total_tokens + self.candidate.usage.total_tokens
333
+
334
+ def side(self, variant: ExperimentVariant) -> VariantExecutionSummary:
335
+ """Return one side's summary."""
336
+ return (
337
+ self.baseline if variant is ExperimentVariant.BASELINE else self.candidate
338
+ )
339
+
340
+
341
+ def weakest_provenance(provenances: Sequence[CostProvenance]) -> CostProvenance:
342
+ """Return the least-claiming provenance among several."""
343
+ return min(provenances, key=lambda value: _PROVENANCE_STRENGTH[value])
344
+
345
+
346
+ def unavailable_cost(detail: str) -> VariantCost:
347
+ """Return the cost of a variant whose economics nothing reported."""
348
+ return VariantCost(
349
+ cost_usd=None, provenance=CostProvenance.UNAVAILABLE, detail=detail
350
+ )
351
+
352
+
353
+ #: What this build says when it has no economics at all. Stated once so the
354
+ #: sentence a reader meets is the same everywhere, and so the day a price feed
355
+ #: lands there is one place that stops being true.
356
+ NO_COST_SOURCE: Final = (
357
+ "no cost figure was reported for this variant, and this build pins no "
358
+ "price to compute one from"
359
+ )
360
+
361
+
362
+ # ---------------------------------------------------------------------------
363
+ # Building
364
+ # ---------------------------------------------------------------------------
365
+
366
+
367
+ def build_comparison_execution_record(
368
+ *,
369
+ run_id: str,
370
+ campaign_spec_digest: Digest,
371
+ campaign_max_concurrent: int,
372
+ execution: RealExecutionResult,
373
+ run_root: Path,
374
+ costs: Mapping[VariantName, VariantCost] | None = None,
375
+ ) -> ComparisonExecutionRecord:
376
+ """Assemble one comparison's operational record from what the run recorded.
377
+
378
+ Every value is read from something the run already wrote: the child
379
+ outcomes, the normalized episodes, the artifact references beside them,
380
+ and the scheduler's own children record. ``costs`` is the seam a price
381
+ feed arrives through; with nothing passed, both sides report an
382
+ unavailable cost, which is what this build's runs honestly produce.
383
+ """
384
+ skew_seconds, first_launched = read_children_record(run_root)
385
+ sides = {
386
+ variant: _summary(
387
+ result=_side(execution, variant),
388
+ campaign_max_concurrent=campaign_max_concurrent,
389
+ schedule=execution.schedule,
390
+ variant=variant,
391
+ cost=(costs or {}).get(variant),
392
+ )
393
+ for variant in _VARIANT_ORDER
394
+ }
395
+ baseline = sides[VariantName.BASELINE]
396
+ candidate = sides[VariantName.CANDIDATE]
397
+
398
+ started_at = min(baseline.started_at, candidate.started_at)
399
+ finished_at = max(baseline.finished_at, candidate.finished_at)
400
+ return ComparisonExecutionRecord(
401
+ schema_version=COMPARISON_EXECUTION_SCHEMA_VERSION,
402
+ run_id=run_id,
403
+ campaign_spec_digest=campaign_spec_digest,
404
+ engine_digest=execution.engine_digest,
405
+ execution_backend=execution.execution_backend,
406
+ schedule=execution.schedule,
407
+ started_at=started_at,
408
+ finished_at=finished_at,
409
+ elapsed_seconds=_seconds(started_at, finished_at),
410
+ launch_skew_seconds=skew_seconds,
411
+ first_launched=first_launched,
412
+ overlap_seconds=_overlap(baseline, candidate),
413
+ campaign_max_concurrent=campaign_max_concurrent,
414
+ outcome=_outcome(baseline, candidate),
415
+ baseline=baseline,
416
+ candidate=candidate,
417
+ )
418
+
419
+
420
+ def read_execution_record(bundle_dir: Path) -> ComparisonExecutionRecord | None:
421
+ """Return the record one proof bundle carries, or ``None`` when it has none.
422
+
423
+ Reading is deliberately separate from checking. The bundle verifier is
424
+ what decides whether a record is signed, intact and about this run; this
425
+ is what a renderer calls afterwards, and it returns nothing rather than
426
+ raising, because a result whose economics cannot be read is still a
427
+ result. A bundle whose record does not verify is already reported by the
428
+ verification the caller ran before it got here.
429
+ """
430
+ path = bundle_dir / EXECUTION_RECORD_FILENAME
431
+ try:
432
+ raw = path.read_bytes()
433
+ except OSError:
434
+ return None
435
+ try:
436
+ envelope = ObjectEnvelope[ComparisonExecutionRecord].model_validate_json(raw)
437
+ except PydanticValidationError:
438
+ return None
439
+ return envelope.payload
440
+
441
+
442
+ def read_children_record(
443
+ run_root: Path,
444
+ ) -> tuple[float | None, ExperimentVariant | None]:
445
+ """Return the launch skew and which side went first, if the run recorded them.
446
+
447
+ The scheduler writes this while the children are being started, which is
448
+ the only moment either value can be observed. A run without the file — a
449
+ sequential schedule records no skew, and an older run may have written
450
+ none — reports neither rather than reconstructing one from timestamps that
451
+ were taken for something else.
452
+ """
453
+ path = children_record_path(run_root)
454
+ try:
455
+ document = json.loads(path.read_bytes())
456
+ except (OSError, ValueError):
457
+ return None, None
458
+ if not isinstance(document, dict):
459
+ return None, None
460
+
461
+ skew = document.get("launch_skew_seconds")
462
+ if not isinstance(skew, int | float) or isinstance(skew, bool) or skew < 0:
463
+ return None, None
464
+
465
+ started = [
466
+ (row.get("started_at"), row.get("variant"))
467
+ for row in document.get("children", [])
468
+ if isinstance(row, dict)
469
+ ]
470
+ first = _first_launched(started)
471
+ if first is None:
472
+ return None, None
473
+ return float(skew), first
474
+
475
+
476
+ # ---------------------------------------------------------------------------
477
+ # The pieces
478
+ # ---------------------------------------------------------------------------
479
+
480
+
481
+ def _summary(
482
+ *,
483
+ result: VariantExecutionResult,
484
+ campaign_max_concurrent: int,
485
+ schedule: VariantSchedule,
486
+ variant: VariantName,
487
+ cost: VariantCost | None,
488
+ ) -> VariantExecutionSummary:
489
+ """Describe one side from its own recorded evidence."""
490
+ outcome = result.child_outcome
491
+ baseline_permits, candidate_permits = divide_concurrency(
492
+ schedule, campaign_max_concurrent
493
+ )
494
+ return VariantExecutionSummary(
495
+ variant=_protocol_variant(variant),
496
+ started_at=outcome.started_at,
497
+ finished_at=outcome.finished_at,
498
+ elapsed_seconds=_seconds(outcome.started_at, outcome.finished_at),
499
+ exit_code=outcome.exit_code,
500
+ cancelled=outcome.cancelled,
501
+ episode_count=len(result.episodes),
502
+ max_concurrent=(
503
+ baseline_permits if variant is VariantName.BASELINE else candidate_permits
504
+ ),
505
+ usage=_usage(result),
506
+ cost=cost or unavailable_cost(NO_COST_SOURCE),
507
+ experiment_manifest_digest=result.experiment_manifest_digest,
508
+ argv_digest=outcome.argv_digest,
509
+ normalized_episodes_digest=result.normalized_episodes.digest,
510
+ raw_traces_digest=result.raw_traces.digest,
511
+ resolved_config_digest=result.resolved_verifiers_config.digest,
512
+ )
513
+
514
+
515
+ def _usage(result: VariantExecutionResult) -> VariantUsage:
516
+ """Sum one side's consumption from the engine's normalized traces.
517
+
518
+ Two measurements with two different availabilities, kept apart. Model
519
+ calls are counted by every trace, so they are known whenever this variant
520
+ recorded a trace at all. Token usage is reported per trace and may be
521
+ absent, so a trace that reports none is *counted* rather than read as a
522
+ zero, and the coverage travels with the totals.
523
+ """
524
+ traces_total = 0
525
+ traces_with_usage = 0
526
+ model_calls = 0
527
+ totals = {"input": 0, "cached": 0, "output": 0, "total": 0}
528
+ for episode in result.episodes:
529
+ for trace in episode.traces:
530
+ traces_total += 1
531
+ model_calls += trace.model_calls
532
+ usage = trace.usage
533
+ if usage is None:
534
+ continue
535
+ traces_with_usage += 1
536
+ totals["input"] += usage.input_tokens
537
+ totals["cached"] += usage.cached_input_tokens or 0
538
+ totals["output"] += usage.output_tokens
539
+ totals["total"] += usage.total_tokens
540
+
541
+ counted = None if traces_total == 0 else model_calls
542
+ if traces_with_usage == 0:
543
+ return VariantUsage(
544
+ provenance=UsageProvenance.UNAVAILABLE,
545
+ model_calls=counted,
546
+ traces_total=traces_total,
547
+ traces_with_usage=0,
548
+ )
549
+ return VariantUsage(
550
+ provenance=UsageProvenance.NORMALIZED_TRACES,
551
+ model_calls=counted,
552
+ input_tokens=totals["input"],
553
+ cached_input_tokens=totals["cached"],
554
+ output_tokens=totals["output"],
555
+ total_tokens=totals["total"],
556
+ traces_total=traces_total,
557
+ traces_with_usage=traces_with_usage,
558
+ )
559
+
560
+
561
+ def _overlap(
562
+ baseline: VariantExecutionSummary, candidate: VariantExecutionSummary
563
+ ) -> float:
564
+ """Return how long both sides were running at once.
565
+
566
+ A parallel comparison's whole claim is that the two sides met the same
567
+ conditions, and the overlap is the measurable part of it. A sequential
568
+ schedule produces zero here, which is the true answer rather than a
569
+ missing one.
570
+ """
571
+ start = max(baseline.started_at, candidate.started_at)
572
+ finish = min(baseline.finished_at, candidate.finished_at)
573
+ return _seconds(start, finish) if finish > start else 0.0
574
+
575
+
576
+ def _outcome(
577
+ baseline: VariantExecutionSummary, candidate: VariantExecutionSummary
578
+ ) -> PairOutcome:
579
+ """Say how the pair ended, from what the children themselves recorded."""
580
+ if baseline.cancelled or candidate.cancelled:
581
+ return PairOutcome.CANCELLED
582
+ if baseline.exit_code != 0 or candidate.exit_code != 0:
583
+ return PairOutcome.FAILED
584
+ return PairOutcome.COMPLETED
585
+
586
+
587
+ def _first_launched(
588
+ started: Sequence[tuple[object, object]],
589
+ ) -> ExperimentVariant | None:
590
+ """Return which variant the operational record started first."""
591
+ parsed: list[tuple[datetime, ExperimentVariant]] = []
592
+ for stamp, variant in started:
593
+ if not isinstance(stamp, str) or not isinstance(variant, str):
594
+ return None
595
+ try:
596
+ when = datetime.fromisoformat(stamp)
597
+ side = ExperimentVariant(variant)
598
+ except ValueError:
599
+ return None
600
+ parsed.append((when, side))
601
+ if len(parsed) != len(_VARIANT_ORDER):
602
+ return None
603
+ return min(parsed, key=lambda item: item[0])[1]
604
+
605
+
606
+ def _seconds(start: datetime, finish: datetime) -> float:
607
+ """Return a non-negative duration between two recorded instants."""
608
+ return max(0.0, (finish - start).total_seconds())
609
+
610
+
611
+ def _side(
612
+ execution: RealExecutionResult, variant: VariantName
613
+ ) -> VariantExecutionResult:
614
+ """Return one side of an execution result."""
615
+ return (
616
+ execution.baseline if variant is VariantName.BASELINE else execution.candidate
617
+ )
618
+
619
+
620
+ def _protocol_variant(variant: VariantName) -> ExperimentVariant:
621
+ """Return the protocol spelling of one variant name."""
622
+ return (
623
+ ExperimentVariant.BASELINE
624
+ if variant is VariantName.BASELINE
625
+ else ExperimentVariant.CANDIDATE
626
+ )
627
+
628
+
629
+ type SignedComparisonExecutionRecord = ObjectEnvelope[ComparisonExecutionRecord]
630
+ """One record inside the envelope the executor signed it with."""