techtree 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. techtree/__init__.py +35 -0
  2. techtree/__main__.py +14 -0
  3. techtree/canonical.py +239 -0
  4. techtree/catalog/__init__.py +25 -0
  5. techtree/catalog/repository.py +400 -0
  6. techtree/catalog/service.py +419 -0
  7. techtree/cli/__init__.py +1 -0
  8. techtree/cli/app.py +416 -0
  9. techtree/cli/commands/__init__.py +1 -0
  10. techtree/cli/commands/climb.py +1223 -0
  11. techtree/cli/commands/doctor.py +147 -0
  12. techtree/cli/commands/engine.py +207 -0
  13. techtree/cli/commands/proof.py +556 -0
  14. techtree/cli/commands/publish.py +447 -0
  15. techtree/cli/commands/release.py +303 -0
  16. techtree/cli/commands/run.py +1067 -0
  17. techtree/cli/commands/setup.py +181 -0
  18. techtree/cli/commands/skill.py +221 -0
  19. techtree/cli/commands/uplift.py +698 -0
  20. techtree/cli/commands/withdraw.py +212 -0
  21. techtree/cli/confirm.py +47 -0
  22. techtree/cli/context.py +96 -0
  23. techtree/cli/invoke.py +220 -0
  24. techtree/cli/output.py +280 -0
  25. techtree/constants.py +138 -0
  26. techtree/crypto.py +128 -0
  27. techtree/doctor/__init__.py +1 -0
  28. techtree/doctor/checks.py +675 -0
  29. techtree/doctor/execution_checks.py +435 -0
  30. techtree/doctor/service.py +326 -0
  31. techtree/drafts/__init__.py +32 -0
  32. techtree/drafts/source.py +146 -0
  33. techtree/drafts/store.py +992 -0
  34. techtree/engines/__init__.py +1 -0
  35. techtree/engines/bundle.py +251 -0
  36. techtree/engines/installer.py +679 -0
  37. techtree/engines/registry.py +235 -0
  38. techtree/engines/runner.py +170 -0
  39. techtree/errors.py +262 -0
  40. techtree/fs.py +234 -0
  41. techtree/harness.py +108 -0
  42. techtree/identity/__init__.py +41 -0
  43. techtree/identity/models.py +113 -0
  44. techtree/identity/service.py +199 -0
  45. techtree/identity/store.py +263 -0
  46. techtree/ids.py +85 -0
  47. techtree/manifests/__init__.py +39 -0
  48. techtree/manifests/builder.py +433 -0
  49. techtree/manifests/compare.py +376 -0
  50. techtree/models/__init__.py +282 -0
  51. techtree/models/base.py +201 -0
  52. techtree/models/campaign.py +484 -0
  53. techtree/models/catalog.py +227 -0
  54. techtree/models/cli.py +151 -0
  55. techtree/models/climb.py +254 -0
  56. techtree/models/data_policy.py +130 -0
  57. techtree/models/engine.py +156 -0
  58. techtree/models/episode_receipt.py +130 -0
  59. techtree/models/evaluation_backend.py +113 -0
  60. techtree/models/experiment.py +154 -0
  61. techtree/models/run.py +214 -0
  62. techtree/models/skill.py +156 -0
  63. techtree/models/uplift_report.py +158 -0
  64. techtree/models/validation.py +299 -0
  65. techtree/paths.py +116 -0
  66. techtree/presentation/__init__.py +31 -0
  67. techtree/presentation/build.py +1242 -0
  68. techtree/presentation/compact.py +246 -0
  69. techtree/presentation/evidence.py +169 -0
  70. techtree/presentation/models.py +358 -0
  71. techtree/presentation/rich.py +312 -0
  72. techtree/presentation/sanitize.py +156 -0
  73. techtree/publication/__init__.py +44 -0
  74. techtree/publication/address.py +180 -0
  75. techtree/publication/coordinates.py +26 -0
  76. techtree/publication/journal.py +212 -0
  77. techtree/publication/keccak.py +183 -0
  78. techtree/publication/models.py +209 -0
  79. techtree/publication/offer.py +35 -0
  80. techtree/publication/service.py +618 -0
  81. techtree/publication/transport.py +296 -0
  82. techtree/publication/verify.py +242 -0
  83. techtree/publication/withdraw.py +156 -0
  84. techtree/py.typed +0 -0
  85. techtree/receipts/__init__.py +52 -0
  86. techtree/receipts/bundle.py +578 -0
  87. techtree/receipts/compare.py +1065 -0
  88. techtree/receipts/episode.py +672 -0
  89. techtree/receipts/execution.py +630 -0
  90. techtree/receipts/observed.py +474 -0
  91. techtree/receipts/set.py +336 -0
  92. techtree/receipts/uplift.py +655 -0
  93. techtree/receipts/verify.py +1055 -0
  94. techtree/release/__init__.py +9 -0
  95. techtree/release/bootstrap.py +509 -0
  96. techtree/release/checks.py +376 -0
  97. techtree/release/document.py +125 -0
  98. techtree/release/generate.py +221 -0
  99. techtree/release/models.py +293 -0
  100. techtree/release/provenance.py +109 -0
  101. techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
  102. techtree/resources/catalog/catalog.json +32 -0
  103. techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
  104. techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
  105. techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
  106. techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
  107. techtree/resources/engines/default/engine.json +20 -0
  108. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
  109. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
  110. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
  111. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
  112. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
  113. techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
  114. techtree/resources/engines/default/pyproject.toml +23 -0
  115. techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
  116. techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
  117. techtree/resources/engines/default/tools/normalize_validation.py +222 -0
  118. techtree/resources/engines/default/uv.lock +1758 -0
  119. techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
  120. techtree/resources/release/build-provenance.json +4 -0
  121. techtree/resources/release/release-core.json +24 -0
  122. techtree/runs/__init__.py +31 -0
  123. techtree/runs/artifacts.py +750 -0
  124. techtree/runs/child_registry.py +228 -0
  125. techtree/runs/events.py +478 -0
  126. techtree/runs/executor.py +140 -0
  127. techtree/runs/fake.py +741 -0
  128. techtree/runs/launcher.py +253 -0
  129. techtree/runs/machine.py +489 -0
  130. techtree/runs/real.py +789 -0
  131. techtree/runs/service.py +616 -0
  132. techtree/runs/store.py +555 -0
  133. techtree/runs/validation.py +259 -0
  134. techtree/runs/variants.py +684 -0
  135. techtree/settings.py +143 -0
  136. techtree/skills/__init__.py +14 -0
  137. techtree/skills/archive.py +282 -0
  138. techtree/skills/policy.py +62 -0
  139. techtree/skills/scanner.py +394 -0
  140. techtree/skills/service.py +752 -0
  141. techtree/skills/starter.py +434 -0
  142. techtree/tasksets/__init__.py +1 -0
  143. techtree/tasksets/membership.py +269 -0
  144. techtree/tasksets/provider.py +207 -0
  145. techtree/tasksets/resolver.py +311 -0
  146. techtree/tasksets/service.py +484 -0
  147. techtree/tasksets/verifiers_cli.py +538 -0
  148. techtree/uplift/__init__.py +20 -0
  149. techtree/uplift/context.py +544 -0
  150. techtree/uplift/derive.py +203 -0
  151. techtree/uplift/public_tasks.py +151 -0
  152. techtree/uplift/service.py +719 -0
  153. techtree/uplift/source.py +160 -0
  154. techtree/verifiers/__init__.py +31 -0
  155. techtree/verifiers/budget.py +219 -0
  156. techtree/verifiers/child.py +633 -0
  157. techtree/verifiers/compiler.py +432 -0
  158. techtree/verifiers/config.py +365 -0
  159. techtree/verifiers/credentials.py +321 -0
  160. techtree/verifiers/image.py +126 -0
  161. techtree/verifiers/models.py +527 -0
  162. techtree/verifiers/outputs.py +368 -0
  163. techtree/verifiers/progress.py +192 -0
  164. techtree/verifiers/supervisor.py +341 -0
  165. techtree/verifiers/verify.py +782 -0
  166. techtree/version.py +39 -0
  167. techtree/worker/__init__.py +18 -0
  168. techtree/worker/execute.py +487 -0
  169. techtree/worker/main.py +57 -0
  170. techtree-0.1.0.dist-info/METADATA +344 -0
  171. techtree-0.1.0.dist-info/RECORD +174 -0
  172. techtree-0.1.0.dist-info/WHEEL +4 -0
  173. techtree-0.1.0.dist-info/entry_points.txt +3 -0
  174. techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,684 @@
1
+ """Running both sides of a comparison at once. Spec section 6.15.
2
+
3
+ A controlled comparison is only as good as the conditions the two variants met,
4
+ and the conditions a provider offers drift: queue depth, routing, and a
5
+ model's own revision are not constant over the forty minutes an agentic
6
+ taskset takes. Running the variants side by side is how that drift is shared
7
+ instead of assigned to whichever side went second, and it is why the start
8
+ barrier in this module is a scientific control rather than a performance
9
+ optimisation.
10
+
11
+ Three rules follow from it.
12
+
13
+ *Nothing is verified after the first launch.* Every input either variant needs
14
+ is checked before either child starts, because a missing candidate config
15
+ discovered after the baseline is already talking to a provider costs the run
16
+ its money and its comparability at once.
17
+
18
+ *Nothing is written between the two launches.* The children are started back to
19
+ back and the events that announce them are appended afterwards, so the recorded
20
+ skew measures two ``fork`` calls rather than two ``fork`` calls plus a durable
21
+ append to a journal.
22
+
23
+ *One variant's failure ends the other.* A pair is the unit of a comparison. A
24
+ baseline that finished cannot be reported against a candidate that did not, so
25
+ a failed child causes its sibling to be terminated and the whole pair to fail —
26
+ with both children's partial evidence left exactly where they wrote it, because
27
+ a failed run is still the only record of what happened.
28
+
29
+ Cancellation is the same shape and a different meaning: the sibling is
30
+ terminated for the same reason, and the run produced no answer rather than a
31
+ wrong one.
32
+ """
33
+
34
+ from __future__ import annotations
35
+
36
+ import contextlib
37
+ import time
38
+ from collections.abc import Callable, Iterator
39
+ from dataclasses import dataclass
40
+ from datetime import UTC, datetime
41
+ from pathlib import Path
42
+ from typing import Final
43
+
44
+ from techtree.errors import (
45
+ CancellationError,
46
+ RunError,
47
+ TechtreeError,
48
+ ValidationError,
49
+ )
50
+ from techtree.models.base import JsonValue
51
+ from techtree.models.campaign import VariantSchedule
52
+ from techtree.models.run import RunPhase, VariantProgress
53
+ from techtree.runs.child_registry import (
54
+ ChildRegistry,
55
+ EvaluationChild,
56
+ LaunchedChild,
57
+ write_children_record,
58
+ )
59
+ from techtree.runs.events import (
60
+ DETAIL_COMPLETED,
61
+ DETAIL_CURRENT,
62
+ DETAIL_ERRORED,
63
+ DETAIL_LABEL,
64
+ DETAIL_RUNNING,
65
+ DETAIL_STATE,
66
+ DETAIL_TOTAL,
67
+ DETAIL_VARIANT,
68
+ PROGRESS_UPDATED,
69
+ VARIANT_COMPLETED,
70
+ VARIANT_PROGRESS,
71
+ VARIANT_STARTED,
72
+ )
73
+ from techtree.runs.executor import raise_if_cancel_requested
74
+ from techtree.runs.store import RunStore
75
+ from techtree.verifiers.child import DEFAULT_GRACE_SECONDS
76
+ from techtree.verifiers.models import (
77
+ ChildProcessOutcome,
78
+ VariantExecutionPlan,
79
+ VariantName,
80
+ )
81
+ from techtree.verifiers.outputs import TRACES_FILENAME
82
+ from techtree.verifiers.progress import inspect_progress, pending_progress
83
+
84
+ __all__ = [
85
+ "DEFAULT_POLL_INTERVAL_SECONDS",
86
+ "VARIANT_CHILD_START_FAILED",
87
+ "VARIANT_CONCURRENCY_EXCEEDED",
88
+ "VARIANT_EXECUTION_FAILED",
89
+ "VARIANT_INPUTS_MISSING",
90
+ "LaunchSkew",
91
+ "VariantPair",
92
+ "VariantPairOutcome",
93
+ "VariantScheduler",
94
+ "require_concurrency_budget",
95
+ ]
96
+
97
+ #: Stable error codes.
98
+ VARIANT_INPUTS_MISSING: Final = "variant_inputs_missing"
99
+ VARIANT_CHILD_START_FAILED: Final = "variant_child_start_failed"
100
+ VARIANT_EXECUTION_FAILED: Final = "variant_execution_failed"
101
+ VARIANT_CONCURRENCY_EXCEEDED: Final = "variant_concurrency_exceeded"
102
+
103
+ #: How often the scheduler reads both variants' evidence. Spec section 6.15.
104
+ DEFAULT_POLL_INTERVAL_SECONDS: Final = 0.25
105
+
106
+ #: Variants are always addressed in comparison order, never in completion or
107
+ #: start order.
108
+ _VARIANT_ORDER: Final[tuple[VariantName, ...]] = (
109
+ VariantName.BASELINE,
110
+ VariantName.CANDIDATE,
111
+ )
112
+
113
+ #: What a sequential schedule's progress lines are labelled with.
114
+ _EPISODE_LABEL: Final = "{variant} episodes"
115
+
116
+
117
+ # ---------------------------------------------------------------------------
118
+ # The pair
119
+ # ---------------------------------------------------------------------------
120
+
121
+
122
+ @dataclass(frozen=True)
123
+ class VariantPair:
124
+ """The two plans one comparison executes. Spec section 6.15."""
125
+
126
+ baseline: VariantExecutionPlan
127
+ candidate: VariantExecutionPlan
128
+
129
+ def __post_init__(self) -> None:
130
+ """Reject a pair that is not one comparison of one taskset."""
131
+ if self.baseline.variant is not VariantName.BASELINE:
132
+ raise ValidationError(
133
+ "the baseline slot holds the baseline plan",
134
+ code=VARIANT_INPUTS_MISSING,
135
+ details={"variant": self.baseline.variant.value},
136
+ )
137
+ if self.candidate.variant is not VariantName.CANDIDATE:
138
+ raise ValidationError(
139
+ "the candidate slot holds the candidate plan",
140
+ code=VARIANT_INPUTS_MISSING,
141
+ details={"variant": self.candidate.variant.value},
142
+ )
143
+ if self.baseline.task_count != self.candidate.task_count:
144
+ raise ValidationError(
145
+ "the two variants of a comparison score the same tasks; this "
146
+ f"pair scores {self.baseline.task_count} against "
147
+ f"{self.candidate.task_count}",
148
+ code=VARIANT_INPUTS_MISSING,
149
+ details={
150
+ "baseline_task_count": self.baseline.task_count,
151
+ "candidate_task_count": self.candidate.task_count,
152
+ },
153
+ )
154
+
155
+ def plan(self, variant: VariantName) -> VariantExecutionPlan:
156
+ """Return one side's plan."""
157
+ return self.baseline if variant is VariantName.BASELINE else self.candidate
158
+
159
+ @property
160
+ def task_count(self) -> int:
161
+ """How many tasks each side scores."""
162
+ return self.baseline.task_count
163
+
164
+ @property
165
+ def total_max_concurrent(self) -> int:
166
+ """How many episodes both sides together may have in flight."""
167
+ return self.baseline.max_concurrent + self.candidate.max_concurrent
168
+
169
+
170
+ def require_concurrency_budget(pair: VariantPair, *, max_concurrent: int) -> None:
171
+ """Refuse a pair whose two halves exceed the Campaign's own bound.
172
+
173
+ ``max_concurrent`` is Campaign-wide (spec section 3.2), and dividing it is
174
+ :func:`techtree.verifiers.compiler.divide_concurrency`'s job. This is the
175
+ check that the division actually held: granting each side the full
176
+ allowance would double the live subject count the Campaign declared, and a
177
+ Campaign's concurrency bound is a statement about how much of a provider it
178
+ is willing to occupy at once.
179
+ """
180
+ if pair.total_max_concurrent > max_concurrent:
181
+ raise ValidationError(
182
+ f"this pair would run {pair.total_max_concurrent} episodes at once "
183
+ f"and the Campaign permits {max_concurrent}",
184
+ code=VARIANT_CONCURRENCY_EXCEEDED,
185
+ details={
186
+ "campaign_max_concurrent": max_concurrent,
187
+ "baseline_max_concurrent": pair.baseline.max_concurrent,
188
+ "candidate_max_concurrent": pair.candidate.max_concurrent,
189
+ },
190
+ )
191
+
192
+
193
+ # ---------------------------------------------------------------------------
194
+ # What one pair's execution produced
195
+ # ---------------------------------------------------------------------------
196
+
197
+
198
+ @dataclass(frozen=True)
199
+ class LaunchSkew:
200
+ """How far apart the two children actually started. Spec section 6.15.
201
+
202
+ The timestamps are the parent's observation of each launch, taken the
203
+ instant the child's ``start`` returned. ``seconds`` is measured on the
204
+ monotonic clock instead of by subtracting the two timestamps, so a system
205
+ clock adjustment between the launches cannot produce a negative skew or a
206
+ minute-long one.
207
+ """
208
+
209
+ baseline_started_at: datetime
210
+ candidate_started_at: datetime
211
+ seconds: float
212
+ first: VariantName
213
+
214
+
215
+ @dataclass(frozen=True)
216
+ class VariantPairOutcome:
217
+ """Both children's outcomes, and how far apart they were launched.
218
+
219
+ Spec section 6.15 returns the pair of outcomes; the skew rides with them
220
+ because it is a property of the pair rather than of either child, and
221
+ section 6.15 requires the run to record it.
222
+ """
223
+
224
+ baseline: ChildProcessOutcome
225
+ candidate: ChildProcessOutcome
226
+ schedule: VariantSchedule
227
+ skew: LaunchSkew | None
228
+
229
+ @property
230
+ def outcomes(self) -> tuple[ChildProcessOutcome, ChildProcessOutcome]:
231
+ """Both outcomes, in comparison order."""
232
+ return self.baseline, self.candidate
233
+
234
+
235
+ # ---------------------------------------------------------------------------
236
+ # The scheduler
237
+ # ---------------------------------------------------------------------------
238
+
239
+
240
+ class VariantScheduler:
241
+ """Starts, watches, and stops the children of one comparison."""
242
+
243
+ def __init__(
244
+ self,
245
+ *,
246
+ run_store: RunStore,
247
+ child_registry: ChildRegistry,
248
+ poll_interval_seconds: float = DEFAULT_POLL_INTERVAL_SECONDS,
249
+ grace_seconds: float = DEFAULT_GRACE_SECONDS,
250
+ clock: Callable[[], datetime] | None = None,
251
+ ) -> None:
252
+ if poll_interval_seconds <= 0:
253
+ raise ValidationError(
254
+ "a poller needs a positive interval",
255
+ details={"poll_interval_seconds": poll_interval_seconds},
256
+ )
257
+ self._run_store = run_store
258
+ self._children = child_registry
259
+ self._poll_interval = poll_interval_seconds
260
+ self._grace = grace_seconds
261
+ self._clock = clock or _utc_now
262
+
263
+ # -- parallel ----------------------------------------------------------
264
+
265
+ def execute_parallel(
266
+ self,
267
+ *,
268
+ run_id: str,
269
+ run_root: Path,
270
+ pair: VariantPair,
271
+ baseline_child: EvaluationChild,
272
+ candidate_child: EvaluationChild,
273
+ ) -> VariantPairOutcome:
274
+ """Run both variants side by side under one ``running_variants`` phase.
275
+
276
+ Both children are started before either is polled. If the second cannot
277
+ be started the first is stopped again, because a comparison with one
278
+ live side is not a comparison and the money it would spend buys
279
+ nothing.
280
+ """
281
+ children = {
282
+ VariantName.BASELINE: baseline_child,
283
+ VariantName.CANDIDATE: candidate_child,
284
+ }
285
+ self._require_inputs(pair, children)
286
+ raise_if_cancel_requested(self._run_store, run_id)
287
+ self._run_store.append(run_id, phase=RunPhase.RUNNING_VARIANTS)
288
+
289
+ skew = self._start_both(run_id, run_root=run_root, children=children)
290
+ self._announce(run_id, pair, VariantName.BASELINE)
291
+ self._announce(run_id, pair, VariantName.CANDIDATE)
292
+
293
+ outcomes = self._watch_both(run_id, pair, children)
294
+ return VariantPairOutcome(
295
+ baseline=outcomes[VariantName.BASELINE],
296
+ candidate=outcomes[VariantName.CANDIDATE],
297
+ schedule=VariantSchedule.PARALLEL,
298
+ skew=skew,
299
+ )
300
+
301
+ # -- sequential --------------------------------------------------------
302
+
303
+ def execute_sequential(
304
+ self,
305
+ *,
306
+ run_id: str,
307
+ run_root: Path,
308
+ pair: VariantPair,
309
+ baseline_child: EvaluationChild,
310
+ candidate_child: EvaluationChild,
311
+ ) -> VariantPairOutcome:
312
+ """Run one variant and then the other, for a Campaign that asks for it.
313
+
314
+ The two sequential phases stay what they have always been and no
315
+ variant event is recorded, because ``variant.started`` and its siblings
316
+ say "both sides are in flight" and here they are not. The inputs are
317
+ still checked as a pair before the first child starts: a candidate
318
+ config that does not exist is worth discovering before the baseline is
319
+ paid for, whichever order they run in.
320
+ """
321
+ children = {
322
+ VariantName.BASELINE: baseline_child,
323
+ VariantName.CANDIDATE: candidate_child,
324
+ }
325
+ self._require_inputs(pair, children)
326
+
327
+ outcomes: dict[VariantName, ChildProcessOutcome] = {}
328
+ launched: list[LaunchedChild] = []
329
+ for variant, phase in (
330
+ (VariantName.BASELINE, RunPhase.RUNNING_BASELINE),
331
+ (VariantName.CANDIDATE, RunPhase.RUNNING_CANDIDATE),
332
+ ):
333
+ raise_if_cancel_requested(self._run_store, run_id)
334
+ self._run_store.append(run_id, phase=phase)
335
+ child = children[variant]
336
+ launched.append(self._start_one(run_id, child))
337
+ write_children_record(
338
+ run_root=run_root,
339
+ run_id=run_id,
340
+ schedule=VariantSchedule.SEQUENTIAL,
341
+ children=launched,
342
+ launch_skew_seconds=None,
343
+ )
344
+ outcomes[variant] = self._watch_one(run_id, pair, child)
345
+
346
+ return VariantPairOutcome(
347
+ baseline=outcomes[VariantName.BASELINE],
348
+ candidate=outcomes[VariantName.CANDIDATE],
349
+ schedule=VariantSchedule.SEQUENTIAL,
350
+ skew=None,
351
+ )
352
+
353
+ # -- the start barrier -------------------------------------------------
354
+
355
+ def _require_inputs(
356
+ self,
357
+ pair: VariantPair,
358
+ children: dict[VariantName, EvaluationChild],
359
+ ) -> None:
360
+ """Check every input both variants need, before either child starts."""
361
+ for variant in _VARIANT_ORDER:
362
+ if children[variant].variant is not variant:
363
+ raise ValidationError(
364
+ f"the {variant.value} slot holds a "
365
+ f"{children[variant].variant.value} child",
366
+ code=VARIANT_INPUTS_MISSING,
367
+ details={"variant": variant.value},
368
+ )
369
+ plan = pair.plan(variant)
370
+ for label, path in (
371
+ ("compiled config", Path(plan.verifiers_input_config_path)),
372
+ ("experiment manifest", Path(plan.experiment_manifest_path)),
373
+ ):
374
+ if not path.is_file():
375
+ raise ValidationError(
376
+ f"the {variant.value} variant's {label} is not on disk, "
377
+ "so neither variant may start",
378
+ code=VARIANT_INPUTS_MISSING,
379
+ details={"variant": variant.value, "path": str(path)},
380
+ )
381
+ for skill in plan.skill_paths:
382
+ if not Path(skill).is_dir():
383
+ raise ValidationError(
384
+ f"the {variant.value} variant declares a skill that is "
385
+ "not staged in the run's own input tree",
386
+ code=VARIANT_INPUTS_MISSING,
387
+ details={"variant": variant.value, "path": skill},
388
+ )
389
+ traces = self._traces_path(plan)
390
+ if traces.exists():
391
+ raise ValidationError(
392
+ f"the {variant.value} variant's output directory already "
393
+ "holds evidence; a run writes its own",
394
+ code=VARIANT_INPUTS_MISSING,
395
+ details={"variant": variant.value, "path": str(traces)},
396
+ )
397
+
398
+ # -- starting ----------------------------------------------------------
399
+
400
+ def _start_both(
401
+ self,
402
+ run_id: str,
403
+ *,
404
+ run_root: Path,
405
+ children: dict[VariantName, EvaluationChild],
406
+ ) -> LaunchSkew:
407
+ """Start both children back to back and record how far apart they were."""
408
+ first = self._start_one(run_id, children[VariantName.BASELINE])
409
+ first_monotonic = time.monotonic()
410
+ try:
411
+ second = self._start_one(run_id, children[VariantName.CANDIDATE])
412
+ except BaseException:
413
+ # The pair never existed. Stop the one child that did, so a failed
414
+ # launch does not leave a container talking to a provider.
415
+ self._children.terminate_all(run_id, self._grace)
416
+ raise
417
+ second_monotonic = time.monotonic()
418
+
419
+ skew = LaunchSkew(
420
+ baseline_started_at=first.started_at,
421
+ candidate_started_at=second.started_at,
422
+ seconds=max(second_monotonic - first_monotonic, 0.0),
423
+ first=VariantName.BASELINE,
424
+ )
425
+ write_children_record(
426
+ run_root=run_root,
427
+ run_id=run_id,
428
+ schedule=VariantSchedule.PARALLEL,
429
+ children=[first, second],
430
+ launch_skew_seconds=skew.seconds,
431
+ )
432
+ return skew
433
+
434
+ def _start_one(self, run_id: str, child: EvaluationChild) -> LaunchedChild:
435
+ """Start one child and register it before anything can fail."""
436
+ try:
437
+ child.start()
438
+ except RunError:
439
+ raise
440
+ except OSError as error:
441
+ raise RunError(
442
+ f"the {child.variant.value} evaluation child could not be "
443
+ f"started: {error.strerror or error}",
444
+ code=VARIANT_CHILD_START_FAILED,
445
+ details={"run_id": run_id, "variant": child.variant.value},
446
+ ) from error
447
+ self._children.register(run_id, child)
448
+ return LaunchedChild(
449
+ variant=child.variant,
450
+ pid=child.pid,
451
+ argv_digest=child.argv_digest,
452
+ started_at=self._clock(),
453
+ )
454
+
455
+ # -- watching ----------------------------------------------------------
456
+
457
+ def _watch_both(
458
+ self,
459
+ run_id: str,
460
+ pair: VariantPair,
461
+ children: dict[VariantName, EvaluationChild],
462
+ ) -> dict[VariantName, ChildProcessOutcome]:
463
+ """Poll both children until both have ended, or until one fails."""
464
+ reported: dict[VariantName, VariantProgress | None] = {
465
+ variant: None for variant in _VARIANT_ORDER
466
+ }
467
+ exits: dict[VariantName, int] = {}
468
+ outcomes: dict[VariantName, ChildProcessOutcome] = {}
469
+
470
+ try:
471
+ while len(outcomes) < len(_VARIANT_ORDER):
472
+ raise_if_cancel_requested(self._run_store, run_id)
473
+ for variant in _VARIANT_ORDER:
474
+ if variant in outcomes:
475
+ continue
476
+ child = children[variant]
477
+ code = child.poll()
478
+ progress = self._inspect(pair, variant, code)
479
+ if code is None:
480
+ self._report(run_id, reported, variant, progress)
481
+ continue
482
+ exits[variant] = code
483
+ outcomes[variant] = child.outcome()
484
+ self._children.unregister(run_id, variant)
485
+ self._emit(run_id, VARIANT_COMPLETED, progress)
486
+ reported[variant] = progress
487
+ if code != 0:
488
+ self._stop_sibling(run_id, variant, children, outcomes)
489
+ if len(outcomes) < len(_VARIANT_ORDER):
490
+ time.sleep(self._poll_interval)
491
+ except CancellationError:
492
+ self._children.terminate_all(run_id, self._grace)
493
+ self._collect_remaining(children, outcomes)
494
+ raise
495
+
496
+ self._require_both_succeeded(run_id, exits)
497
+ return outcomes
498
+
499
+ def _watch_one(
500
+ self,
501
+ run_id: str,
502
+ pair: VariantPair,
503
+ child: EvaluationChild,
504
+ ) -> ChildProcessOutcome:
505
+ """Poll one child to its end, reporting phase progress as it goes."""
506
+ variant = child.variant
507
+ last: int | None = None
508
+ try:
509
+ while True:
510
+ raise_if_cancel_requested(self._run_store, run_id)
511
+ code = child.poll()
512
+ progress = self._inspect(pair, variant, code)
513
+ last = self._report_phase_progress(run_id, variant, progress, last)
514
+ if code is not None:
515
+ break
516
+ time.sleep(self._poll_interval)
517
+ except CancellationError:
518
+ self._children.terminate_all(run_id, self._grace)
519
+ raise
520
+
521
+ outcome = child.outcome()
522
+ self._children.unregister(run_id, variant)
523
+ self._require_both_succeeded(run_id, {variant: outcome.exit_code})
524
+ return outcome
525
+
526
+ def _stop_sibling(
527
+ self,
528
+ run_id: str,
529
+ failed: VariantName,
530
+ children: dict[VariantName, EvaluationChild],
531
+ outcomes: dict[VariantName, ChildProcessOutcome],
532
+ ) -> None:
533
+ """Terminate the other side of a pair whose first side failed."""
534
+ for variant in _VARIANT_ORDER:
535
+ if variant is failed or variant in outcomes:
536
+ continue
537
+ sibling = children[variant]
538
+ sibling.terminate(self._grace)
539
+ outcomes[variant] = sibling.outcome()
540
+ self._children.unregister(run_id, variant)
541
+
542
+ def _collect_remaining(
543
+ self,
544
+ children: dict[VariantName, EvaluationChild],
545
+ outcomes: dict[VariantName, ChildProcessOutcome],
546
+ ) -> None:
547
+ """Describe every child that has not been described yet.
548
+
549
+ Called while unwinding, so a child that cannot describe itself is
550
+ skipped rather than allowed to replace the reason the run is stopping.
551
+ """
552
+ for variant in _VARIANT_ORDER:
553
+ if variant in outcomes:
554
+ continue
555
+ try:
556
+ outcomes[variant] = children[variant].outcome()
557
+ except RunError:
558
+ continue
559
+
560
+ def _require_both_succeeded(
561
+ self, run_id: str, exits: dict[VariantName, int]
562
+ ) -> None:
563
+ """Fail the pair when either side did not finish cleanly."""
564
+ failed = sorted(
565
+ (variant.value, code) for variant, code in exits.items() if code != 0
566
+ )
567
+ if not failed:
568
+ return
569
+ detail: list[JsonValue] = [
570
+ {"variant": variant, "exit_code": code} for variant, code in failed
571
+ ]
572
+ names = ", ".join(variant for variant, _ in failed)
573
+ raise RunError(
574
+ f"the {names} evaluation did not finish; a comparison needs both "
575
+ "sides, so the pair failed and the partial evidence was kept",
576
+ code=VARIANT_EXECUTION_FAILED,
577
+ details={"run_id": run_id, "variants": detail},
578
+ )
579
+
580
+ # -- measurement and events -------------------------------------------
581
+
582
+ def _inspect(
583
+ self,
584
+ pair: VariantPair,
585
+ variant: VariantName,
586
+ exit_code: int | None,
587
+ ) -> VariantProgress:
588
+ """Measure one variant from the evidence its child is writing."""
589
+ plan = pair.plan(variant)
590
+ return inspect_progress(
591
+ variant=variant,
592
+ traces_path=self._traces_path(plan),
593
+ total=plan.task_count,
594
+ child_exit_code=exit_code,
595
+ max_concurrent=plan.max_concurrent,
596
+ )
597
+
598
+ def _traces_path(self, plan: VariantExecutionPlan) -> Path:
599
+ """Where one variant's child appends its finished episodes."""
600
+ return Path(plan.verifiers_output_dir) / TRACES_FILENAME
601
+
602
+ def _announce(self, run_id: str, pair: VariantPair, variant: VariantName) -> None:
603
+ """Record that one side of the comparison is up."""
604
+ self._emit(
605
+ run_id,
606
+ VARIANT_STARTED,
607
+ pending_progress(variant, pair.plan(variant).task_count),
608
+ )
609
+
610
+ def _report(
611
+ self,
612
+ run_id: str,
613
+ reported: dict[VariantName, VariantProgress | None],
614
+ variant: VariantName,
615
+ progress: VariantProgress,
616
+ ) -> None:
617
+ """Append a progress event only when something actually moved."""
618
+ if reported[variant] == progress:
619
+ return
620
+ self._emit(run_id, VARIANT_PROGRESS, progress)
621
+ reported[variant] = progress
622
+
623
+ def _report_phase_progress(
624
+ self,
625
+ run_id: str,
626
+ variant: VariantName,
627
+ progress: VariantProgress,
628
+ last: int | None,
629
+ ) -> int:
630
+ """Append one sequential phase's progress line when it advances."""
631
+ if progress.completed == last:
632
+ return progress.completed
633
+ with self._cancellation_aware(run_id):
634
+ self._run_store.append(
635
+ run_id,
636
+ phase=None,
637
+ kind=PROGRESS_UPDATED,
638
+ details={
639
+ DETAIL_CURRENT: progress.completed,
640
+ DETAIL_TOTAL: progress.total,
641
+ DETAIL_LABEL: _EPISODE_LABEL.format(variant=variant.value),
642
+ },
643
+ )
644
+ return progress.completed
645
+
646
+ def _emit(self, run_id: str, kind: str, progress: VariantProgress) -> None:
647
+ """Append one variant event against whatever phase the run is in."""
648
+ with self._cancellation_aware(run_id):
649
+ self._run_store.append(
650
+ run_id,
651
+ phase=None,
652
+ kind=kind,
653
+ details={
654
+ DETAIL_VARIANT: progress.variant,
655
+ DETAIL_COMPLETED: progress.completed,
656
+ DETAIL_TOTAL: progress.total,
657
+ DETAIL_RUNNING: progress.running,
658
+ DETAIL_ERRORED: progress.errored,
659
+ DETAIL_STATE: progress.state,
660
+ },
661
+ )
662
+
663
+ @contextlib.contextmanager
664
+ def _cancellation_aware(self, run_id: str) -> Iterator[None]:
665
+ """Report a refused append as the cancellation that caused it.
666
+
667
+ A variant event belongs to ``running_variants`` and to no other phase,
668
+ so an append that lands after another process asked the run to stop is
669
+ refused by the run's own state machine. That refusal is not a defect
670
+ and it is not what the caller has to unwind for: between the poller's
671
+ cancellation check and its append, another process moved the run to
672
+ ``cancel_requested``, and the cancellation is the fact. Anything else
673
+ that made the append fail is re-raised unchanged.
674
+ """
675
+ try:
676
+ yield
677
+ except TechtreeError:
678
+ raise_if_cancel_requested(self._run_store, run_id)
679
+ raise
680
+
681
+
682
+ def _utc_now() -> datetime:
683
+ """Return the current instant in UTC."""
684
+ return datetime.now(UTC)