techtree 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- techtree/__init__.py +35 -0
- techtree/__main__.py +14 -0
- techtree/canonical.py +239 -0
- techtree/catalog/__init__.py +25 -0
- techtree/catalog/repository.py +400 -0
- techtree/catalog/service.py +419 -0
- techtree/cli/__init__.py +1 -0
- techtree/cli/app.py +416 -0
- techtree/cli/commands/__init__.py +1 -0
- techtree/cli/commands/climb.py +1223 -0
- techtree/cli/commands/doctor.py +147 -0
- techtree/cli/commands/engine.py +207 -0
- techtree/cli/commands/proof.py +556 -0
- techtree/cli/commands/publish.py +447 -0
- techtree/cli/commands/release.py +303 -0
- techtree/cli/commands/run.py +1067 -0
- techtree/cli/commands/setup.py +181 -0
- techtree/cli/commands/skill.py +221 -0
- techtree/cli/commands/uplift.py +698 -0
- techtree/cli/commands/withdraw.py +212 -0
- techtree/cli/confirm.py +47 -0
- techtree/cli/context.py +96 -0
- techtree/cli/invoke.py +220 -0
- techtree/cli/output.py +280 -0
- techtree/constants.py +138 -0
- techtree/crypto.py +128 -0
- techtree/doctor/__init__.py +1 -0
- techtree/doctor/checks.py +675 -0
- techtree/doctor/execution_checks.py +435 -0
- techtree/doctor/service.py +326 -0
- techtree/drafts/__init__.py +32 -0
- techtree/drafts/source.py +146 -0
- techtree/drafts/store.py +992 -0
- techtree/engines/__init__.py +1 -0
- techtree/engines/bundle.py +251 -0
- techtree/engines/installer.py +679 -0
- techtree/engines/registry.py +235 -0
- techtree/engines/runner.py +170 -0
- techtree/errors.py +262 -0
- techtree/fs.py +234 -0
- techtree/harness.py +108 -0
- techtree/identity/__init__.py +41 -0
- techtree/identity/models.py +113 -0
- techtree/identity/service.py +199 -0
- techtree/identity/store.py +263 -0
- techtree/ids.py +85 -0
- techtree/manifests/__init__.py +39 -0
- techtree/manifests/builder.py +433 -0
- techtree/manifests/compare.py +376 -0
- techtree/models/__init__.py +282 -0
- techtree/models/base.py +201 -0
- techtree/models/campaign.py +484 -0
- techtree/models/catalog.py +227 -0
- techtree/models/cli.py +151 -0
- techtree/models/climb.py +254 -0
- techtree/models/data_policy.py +130 -0
- techtree/models/engine.py +156 -0
- techtree/models/episode_receipt.py +130 -0
- techtree/models/evaluation_backend.py +113 -0
- techtree/models/experiment.py +154 -0
- techtree/models/run.py +214 -0
- techtree/models/skill.py +156 -0
- techtree/models/uplift_report.py +158 -0
- techtree/models/validation.py +299 -0
- techtree/paths.py +116 -0
- techtree/presentation/__init__.py +31 -0
- techtree/presentation/build.py +1242 -0
- techtree/presentation/compact.py +246 -0
- techtree/presentation/evidence.py +169 -0
- techtree/presentation/models.py +358 -0
- techtree/presentation/rich.py +312 -0
- techtree/presentation/sanitize.py +156 -0
- techtree/publication/__init__.py +44 -0
- techtree/publication/address.py +180 -0
- techtree/publication/coordinates.py +26 -0
- techtree/publication/journal.py +212 -0
- techtree/publication/keccak.py +183 -0
- techtree/publication/models.py +209 -0
- techtree/publication/offer.py +35 -0
- techtree/publication/service.py +618 -0
- techtree/publication/transport.py +296 -0
- techtree/publication/verify.py +242 -0
- techtree/publication/withdraw.py +156 -0
- techtree/py.typed +0 -0
- techtree/receipts/__init__.py +52 -0
- techtree/receipts/bundle.py +578 -0
- techtree/receipts/compare.py +1065 -0
- techtree/receipts/episode.py +672 -0
- techtree/receipts/execution.py +630 -0
- techtree/receipts/observed.py +474 -0
- techtree/receipts/set.py +336 -0
- techtree/receipts/uplift.py +655 -0
- techtree/receipts/verify.py +1055 -0
- techtree/release/__init__.py +9 -0
- techtree/release/bootstrap.py +509 -0
- techtree/release/checks.py +376 -0
- techtree/release/document.py +125 -0
- techtree/release/generate.py +221 -0
- techtree/release/models.py +293 -0
- techtree/release/provenance.py +109 -0
- techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
- techtree/resources/catalog/catalog.json +32 -0
- techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
- techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
- techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
- techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
- techtree/resources/engines/default/engine.json +20 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
- techtree/resources/engines/default/pyproject.toml +23 -0
- techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
- techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
- techtree/resources/engines/default/tools/normalize_validation.py +222 -0
- techtree/resources/engines/default/uv.lock +1758 -0
- techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
- techtree/resources/release/build-provenance.json +4 -0
- techtree/resources/release/release-core.json +24 -0
- techtree/runs/__init__.py +31 -0
- techtree/runs/artifacts.py +750 -0
- techtree/runs/child_registry.py +228 -0
- techtree/runs/events.py +478 -0
- techtree/runs/executor.py +140 -0
- techtree/runs/fake.py +741 -0
- techtree/runs/launcher.py +253 -0
- techtree/runs/machine.py +489 -0
- techtree/runs/real.py +789 -0
- techtree/runs/service.py +616 -0
- techtree/runs/store.py +555 -0
- techtree/runs/validation.py +259 -0
- techtree/runs/variants.py +684 -0
- techtree/settings.py +143 -0
- techtree/skills/__init__.py +14 -0
- techtree/skills/archive.py +282 -0
- techtree/skills/policy.py +62 -0
- techtree/skills/scanner.py +394 -0
- techtree/skills/service.py +752 -0
- techtree/skills/starter.py +434 -0
- techtree/tasksets/__init__.py +1 -0
- techtree/tasksets/membership.py +269 -0
- techtree/tasksets/provider.py +207 -0
- techtree/tasksets/resolver.py +311 -0
- techtree/tasksets/service.py +484 -0
- techtree/tasksets/verifiers_cli.py +538 -0
- techtree/uplift/__init__.py +20 -0
- techtree/uplift/context.py +544 -0
- techtree/uplift/derive.py +203 -0
- techtree/uplift/public_tasks.py +151 -0
- techtree/uplift/service.py +719 -0
- techtree/uplift/source.py +160 -0
- techtree/verifiers/__init__.py +31 -0
- techtree/verifiers/budget.py +219 -0
- techtree/verifiers/child.py +633 -0
- techtree/verifiers/compiler.py +432 -0
- techtree/verifiers/config.py +365 -0
- techtree/verifiers/credentials.py +321 -0
- techtree/verifiers/image.py +126 -0
- techtree/verifiers/models.py +527 -0
- techtree/verifiers/outputs.py +368 -0
- techtree/verifiers/progress.py +192 -0
- techtree/verifiers/supervisor.py +341 -0
- techtree/verifiers/verify.py +782 -0
- techtree/version.py +39 -0
- techtree/worker/__init__.py +18 -0
- techtree/worker/execute.py +487 -0
- techtree/worker/main.py +57 -0
- techtree-0.1.0.dist-info/METADATA +344 -0
- techtree-0.1.0.dist-info/RECORD +174 -0
- techtree-0.1.0.dist-info/WHEEL +4 -0
- techtree-0.1.0.dist-info/entry_points.txt +3 -0
- techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,684 @@
|
|
|
1
|
+
"""Running both sides of a comparison at once. Spec section 6.15.
|
|
2
|
+
|
|
3
|
+
A controlled comparison is only as good as the conditions the two variants met,
|
|
4
|
+
and the conditions a provider offers drift: queue depth, routing, and a
|
|
5
|
+
model's own revision are not constant over the forty minutes an agentic
|
|
6
|
+
taskset takes. Running the variants side by side is how that drift is shared
|
|
7
|
+
instead of assigned to whichever side went second, and it is why the start
|
|
8
|
+
barrier in this module is a scientific control rather than a performance
|
|
9
|
+
optimisation.
|
|
10
|
+
|
|
11
|
+
Three rules follow from it.
|
|
12
|
+
|
|
13
|
+
*Nothing is verified after the first launch.* Every input either variant needs
|
|
14
|
+
is checked before either child starts, because a missing candidate config
|
|
15
|
+
discovered after the baseline is already talking to a provider costs the run
|
|
16
|
+
its money and its comparability at once.
|
|
17
|
+
|
|
18
|
+
*Nothing is written between the two launches.* The children are started back to
|
|
19
|
+
back and the events that announce them are appended afterwards, so the recorded
|
|
20
|
+
skew measures two ``fork`` calls rather than two ``fork`` calls plus a durable
|
|
21
|
+
append to a journal.
|
|
22
|
+
|
|
23
|
+
*One variant's failure ends the other.* A pair is the unit of a comparison. A
|
|
24
|
+
baseline that finished cannot be reported against a candidate that did not, so
|
|
25
|
+
a failed child causes its sibling to be terminated and the whole pair to fail —
|
|
26
|
+
with both children's partial evidence left exactly where they wrote it, because
|
|
27
|
+
a failed run is still the only record of what happened.
|
|
28
|
+
|
|
29
|
+
Cancellation is the same shape and a different meaning: the sibling is
|
|
30
|
+
terminated for the same reason, and the run produced no answer rather than a
|
|
31
|
+
wrong one.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
from __future__ import annotations
|
|
35
|
+
|
|
36
|
+
import contextlib
|
|
37
|
+
import time
|
|
38
|
+
from collections.abc import Callable, Iterator
|
|
39
|
+
from dataclasses import dataclass
|
|
40
|
+
from datetime import UTC, datetime
|
|
41
|
+
from pathlib import Path
|
|
42
|
+
from typing import Final
|
|
43
|
+
|
|
44
|
+
from techtree.errors import (
|
|
45
|
+
CancellationError,
|
|
46
|
+
RunError,
|
|
47
|
+
TechtreeError,
|
|
48
|
+
ValidationError,
|
|
49
|
+
)
|
|
50
|
+
from techtree.models.base import JsonValue
|
|
51
|
+
from techtree.models.campaign import VariantSchedule
|
|
52
|
+
from techtree.models.run import RunPhase, VariantProgress
|
|
53
|
+
from techtree.runs.child_registry import (
|
|
54
|
+
ChildRegistry,
|
|
55
|
+
EvaluationChild,
|
|
56
|
+
LaunchedChild,
|
|
57
|
+
write_children_record,
|
|
58
|
+
)
|
|
59
|
+
from techtree.runs.events import (
|
|
60
|
+
DETAIL_COMPLETED,
|
|
61
|
+
DETAIL_CURRENT,
|
|
62
|
+
DETAIL_ERRORED,
|
|
63
|
+
DETAIL_LABEL,
|
|
64
|
+
DETAIL_RUNNING,
|
|
65
|
+
DETAIL_STATE,
|
|
66
|
+
DETAIL_TOTAL,
|
|
67
|
+
DETAIL_VARIANT,
|
|
68
|
+
PROGRESS_UPDATED,
|
|
69
|
+
VARIANT_COMPLETED,
|
|
70
|
+
VARIANT_PROGRESS,
|
|
71
|
+
VARIANT_STARTED,
|
|
72
|
+
)
|
|
73
|
+
from techtree.runs.executor import raise_if_cancel_requested
|
|
74
|
+
from techtree.runs.store import RunStore
|
|
75
|
+
from techtree.verifiers.child import DEFAULT_GRACE_SECONDS
|
|
76
|
+
from techtree.verifiers.models import (
|
|
77
|
+
ChildProcessOutcome,
|
|
78
|
+
VariantExecutionPlan,
|
|
79
|
+
VariantName,
|
|
80
|
+
)
|
|
81
|
+
from techtree.verifiers.outputs import TRACES_FILENAME
|
|
82
|
+
from techtree.verifiers.progress import inspect_progress, pending_progress
|
|
83
|
+
|
|
84
|
+
__all__ = [
|
|
85
|
+
"DEFAULT_POLL_INTERVAL_SECONDS",
|
|
86
|
+
"VARIANT_CHILD_START_FAILED",
|
|
87
|
+
"VARIANT_CONCURRENCY_EXCEEDED",
|
|
88
|
+
"VARIANT_EXECUTION_FAILED",
|
|
89
|
+
"VARIANT_INPUTS_MISSING",
|
|
90
|
+
"LaunchSkew",
|
|
91
|
+
"VariantPair",
|
|
92
|
+
"VariantPairOutcome",
|
|
93
|
+
"VariantScheduler",
|
|
94
|
+
"require_concurrency_budget",
|
|
95
|
+
]
|
|
96
|
+
|
|
97
|
+
#: Stable error codes.
|
|
98
|
+
VARIANT_INPUTS_MISSING: Final = "variant_inputs_missing"
|
|
99
|
+
VARIANT_CHILD_START_FAILED: Final = "variant_child_start_failed"
|
|
100
|
+
VARIANT_EXECUTION_FAILED: Final = "variant_execution_failed"
|
|
101
|
+
VARIANT_CONCURRENCY_EXCEEDED: Final = "variant_concurrency_exceeded"
|
|
102
|
+
|
|
103
|
+
#: How often the scheduler reads both variants' evidence. Spec section 6.15.
|
|
104
|
+
DEFAULT_POLL_INTERVAL_SECONDS: Final = 0.25
|
|
105
|
+
|
|
106
|
+
#: Variants are always addressed in comparison order, never in completion or
|
|
107
|
+
#: start order.
|
|
108
|
+
_VARIANT_ORDER: Final[tuple[VariantName, ...]] = (
|
|
109
|
+
VariantName.BASELINE,
|
|
110
|
+
VariantName.CANDIDATE,
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
#: What a sequential schedule's progress lines are labelled with.
|
|
114
|
+
_EPISODE_LABEL: Final = "{variant} episodes"
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
# ---------------------------------------------------------------------------
|
|
118
|
+
# The pair
|
|
119
|
+
# ---------------------------------------------------------------------------
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
@dataclass(frozen=True)
|
|
123
|
+
class VariantPair:
|
|
124
|
+
"""The two plans one comparison executes. Spec section 6.15."""
|
|
125
|
+
|
|
126
|
+
baseline: VariantExecutionPlan
|
|
127
|
+
candidate: VariantExecutionPlan
|
|
128
|
+
|
|
129
|
+
def __post_init__(self) -> None:
|
|
130
|
+
"""Reject a pair that is not one comparison of one taskset."""
|
|
131
|
+
if self.baseline.variant is not VariantName.BASELINE:
|
|
132
|
+
raise ValidationError(
|
|
133
|
+
"the baseline slot holds the baseline plan",
|
|
134
|
+
code=VARIANT_INPUTS_MISSING,
|
|
135
|
+
details={"variant": self.baseline.variant.value},
|
|
136
|
+
)
|
|
137
|
+
if self.candidate.variant is not VariantName.CANDIDATE:
|
|
138
|
+
raise ValidationError(
|
|
139
|
+
"the candidate slot holds the candidate plan",
|
|
140
|
+
code=VARIANT_INPUTS_MISSING,
|
|
141
|
+
details={"variant": self.candidate.variant.value},
|
|
142
|
+
)
|
|
143
|
+
if self.baseline.task_count != self.candidate.task_count:
|
|
144
|
+
raise ValidationError(
|
|
145
|
+
"the two variants of a comparison score the same tasks; this "
|
|
146
|
+
f"pair scores {self.baseline.task_count} against "
|
|
147
|
+
f"{self.candidate.task_count}",
|
|
148
|
+
code=VARIANT_INPUTS_MISSING,
|
|
149
|
+
details={
|
|
150
|
+
"baseline_task_count": self.baseline.task_count,
|
|
151
|
+
"candidate_task_count": self.candidate.task_count,
|
|
152
|
+
},
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
def plan(self, variant: VariantName) -> VariantExecutionPlan:
|
|
156
|
+
"""Return one side's plan."""
|
|
157
|
+
return self.baseline if variant is VariantName.BASELINE else self.candidate
|
|
158
|
+
|
|
159
|
+
@property
|
|
160
|
+
def task_count(self) -> int:
|
|
161
|
+
"""How many tasks each side scores."""
|
|
162
|
+
return self.baseline.task_count
|
|
163
|
+
|
|
164
|
+
@property
|
|
165
|
+
def total_max_concurrent(self) -> int:
|
|
166
|
+
"""How many episodes both sides together may have in flight."""
|
|
167
|
+
return self.baseline.max_concurrent + self.candidate.max_concurrent
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def require_concurrency_budget(pair: VariantPair, *, max_concurrent: int) -> None:
|
|
171
|
+
"""Refuse a pair whose two halves exceed the Campaign's own bound.
|
|
172
|
+
|
|
173
|
+
``max_concurrent`` is Campaign-wide (spec section 3.2), and dividing it is
|
|
174
|
+
:func:`techtree.verifiers.compiler.divide_concurrency`'s job. This is the
|
|
175
|
+
check that the division actually held: granting each side the full
|
|
176
|
+
allowance would double the live subject count the Campaign declared, and a
|
|
177
|
+
Campaign's concurrency bound is a statement about how much of a provider it
|
|
178
|
+
is willing to occupy at once.
|
|
179
|
+
"""
|
|
180
|
+
if pair.total_max_concurrent > max_concurrent:
|
|
181
|
+
raise ValidationError(
|
|
182
|
+
f"this pair would run {pair.total_max_concurrent} episodes at once "
|
|
183
|
+
f"and the Campaign permits {max_concurrent}",
|
|
184
|
+
code=VARIANT_CONCURRENCY_EXCEEDED,
|
|
185
|
+
details={
|
|
186
|
+
"campaign_max_concurrent": max_concurrent,
|
|
187
|
+
"baseline_max_concurrent": pair.baseline.max_concurrent,
|
|
188
|
+
"candidate_max_concurrent": pair.candidate.max_concurrent,
|
|
189
|
+
},
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
# ---------------------------------------------------------------------------
|
|
194
|
+
# What one pair's execution produced
|
|
195
|
+
# ---------------------------------------------------------------------------
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
@dataclass(frozen=True)
|
|
199
|
+
class LaunchSkew:
|
|
200
|
+
"""How far apart the two children actually started. Spec section 6.15.
|
|
201
|
+
|
|
202
|
+
The timestamps are the parent's observation of each launch, taken the
|
|
203
|
+
instant the child's ``start`` returned. ``seconds`` is measured on the
|
|
204
|
+
monotonic clock instead of by subtracting the two timestamps, so a system
|
|
205
|
+
clock adjustment between the launches cannot produce a negative skew or a
|
|
206
|
+
minute-long one.
|
|
207
|
+
"""
|
|
208
|
+
|
|
209
|
+
baseline_started_at: datetime
|
|
210
|
+
candidate_started_at: datetime
|
|
211
|
+
seconds: float
|
|
212
|
+
first: VariantName
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
@dataclass(frozen=True)
|
|
216
|
+
class VariantPairOutcome:
|
|
217
|
+
"""Both children's outcomes, and how far apart they were launched.
|
|
218
|
+
|
|
219
|
+
Spec section 6.15 returns the pair of outcomes; the skew rides with them
|
|
220
|
+
because it is a property of the pair rather than of either child, and
|
|
221
|
+
section 6.15 requires the run to record it.
|
|
222
|
+
"""
|
|
223
|
+
|
|
224
|
+
baseline: ChildProcessOutcome
|
|
225
|
+
candidate: ChildProcessOutcome
|
|
226
|
+
schedule: VariantSchedule
|
|
227
|
+
skew: LaunchSkew | None
|
|
228
|
+
|
|
229
|
+
@property
|
|
230
|
+
def outcomes(self) -> tuple[ChildProcessOutcome, ChildProcessOutcome]:
|
|
231
|
+
"""Both outcomes, in comparison order."""
|
|
232
|
+
return self.baseline, self.candidate
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
# ---------------------------------------------------------------------------
|
|
236
|
+
# The scheduler
|
|
237
|
+
# ---------------------------------------------------------------------------
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
class VariantScheduler:
|
|
241
|
+
"""Starts, watches, and stops the children of one comparison."""
|
|
242
|
+
|
|
243
|
+
def __init__(
|
|
244
|
+
self,
|
|
245
|
+
*,
|
|
246
|
+
run_store: RunStore,
|
|
247
|
+
child_registry: ChildRegistry,
|
|
248
|
+
poll_interval_seconds: float = DEFAULT_POLL_INTERVAL_SECONDS,
|
|
249
|
+
grace_seconds: float = DEFAULT_GRACE_SECONDS,
|
|
250
|
+
clock: Callable[[], datetime] | None = None,
|
|
251
|
+
) -> None:
|
|
252
|
+
if poll_interval_seconds <= 0:
|
|
253
|
+
raise ValidationError(
|
|
254
|
+
"a poller needs a positive interval",
|
|
255
|
+
details={"poll_interval_seconds": poll_interval_seconds},
|
|
256
|
+
)
|
|
257
|
+
self._run_store = run_store
|
|
258
|
+
self._children = child_registry
|
|
259
|
+
self._poll_interval = poll_interval_seconds
|
|
260
|
+
self._grace = grace_seconds
|
|
261
|
+
self._clock = clock or _utc_now
|
|
262
|
+
|
|
263
|
+
# -- parallel ----------------------------------------------------------
|
|
264
|
+
|
|
265
|
+
def execute_parallel(
|
|
266
|
+
self,
|
|
267
|
+
*,
|
|
268
|
+
run_id: str,
|
|
269
|
+
run_root: Path,
|
|
270
|
+
pair: VariantPair,
|
|
271
|
+
baseline_child: EvaluationChild,
|
|
272
|
+
candidate_child: EvaluationChild,
|
|
273
|
+
) -> VariantPairOutcome:
|
|
274
|
+
"""Run both variants side by side under one ``running_variants`` phase.
|
|
275
|
+
|
|
276
|
+
Both children are started before either is polled. If the second cannot
|
|
277
|
+
be started the first is stopped again, because a comparison with one
|
|
278
|
+
live side is not a comparison and the money it would spend buys
|
|
279
|
+
nothing.
|
|
280
|
+
"""
|
|
281
|
+
children = {
|
|
282
|
+
VariantName.BASELINE: baseline_child,
|
|
283
|
+
VariantName.CANDIDATE: candidate_child,
|
|
284
|
+
}
|
|
285
|
+
self._require_inputs(pair, children)
|
|
286
|
+
raise_if_cancel_requested(self._run_store, run_id)
|
|
287
|
+
self._run_store.append(run_id, phase=RunPhase.RUNNING_VARIANTS)
|
|
288
|
+
|
|
289
|
+
skew = self._start_both(run_id, run_root=run_root, children=children)
|
|
290
|
+
self._announce(run_id, pair, VariantName.BASELINE)
|
|
291
|
+
self._announce(run_id, pair, VariantName.CANDIDATE)
|
|
292
|
+
|
|
293
|
+
outcomes = self._watch_both(run_id, pair, children)
|
|
294
|
+
return VariantPairOutcome(
|
|
295
|
+
baseline=outcomes[VariantName.BASELINE],
|
|
296
|
+
candidate=outcomes[VariantName.CANDIDATE],
|
|
297
|
+
schedule=VariantSchedule.PARALLEL,
|
|
298
|
+
skew=skew,
|
|
299
|
+
)
|
|
300
|
+
|
|
301
|
+
# -- sequential --------------------------------------------------------
|
|
302
|
+
|
|
303
|
+
def execute_sequential(
|
|
304
|
+
self,
|
|
305
|
+
*,
|
|
306
|
+
run_id: str,
|
|
307
|
+
run_root: Path,
|
|
308
|
+
pair: VariantPair,
|
|
309
|
+
baseline_child: EvaluationChild,
|
|
310
|
+
candidate_child: EvaluationChild,
|
|
311
|
+
) -> VariantPairOutcome:
|
|
312
|
+
"""Run one variant and then the other, for a Campaign that asks for it.
|
|
313
|
+
|
|
314
|
+
The two sequential phases stay what they have always been and no
|
|
315
|
+
variant event is recorded, because ``variant.started`` and its siblings
|
|
316
|
+
say "both sides are in flight" and here they are not. The inputs are
|
|
317
|
+
still checked as a pair before the first child starts: a candidate
|
|
318
|
+
config that does not exist is worth discovering before the baseline is
|
|
319
|
+
paid for, whichever order they run in.
|
|
320
|
+
"""
|
|
321
|
+
children = {
|
|
322
|
+
VariantName.BASELINE: baseline_child,
|
|
323
|
+
VariantName.CANDIDATE: candidate_child,
|
|
324
|
+
}
|
|
325
|
+
self._require_inputs(pair, children)
|
|
326
|
+
|
|
327
|
+
outcomes: dict[VariantName, ChildProcessOutcome] = {}
|
|
328
|
+
launched: list[LaunchedChild] = []
|
|
329
|
+
for variant, phase in (
|
|
330
|
+
(VariantName.BASELINE, RunPhase.RUNNING_BASELINE),
|
|
331
|
+
(VariantName.CANDIDATE, RunPhase.RUNNING_CANDIDATE),
|
|
332
|
+
):
|
|
333
|
+
raise_if_cancel_requested(self._run_store, run_id)
|
|
334
|
+
self._run_store.append(run_id, phase=phase)
|
|
335
|
+
child = children[variant]
|
|
336
|
+
launched.append(self._start_one(run_id, child))
|
|
337
|
+
write_children_record(
|
|
338
|
+
run_root=run_root,
|
|
339
|
+
run_id=run_id,
|
|
340
|
+
schedule=VariantSchedule.SEQUENTIAL,
|
|
341
|
+
children=launched,
|
|
342
|
+
launch_skew_seconds=None,
|
|
343
|
+
)
|
|
344
|
+
outcomes[variant] = self._watch_one(run_id, pair, child)
|
|
345
|
+
|
|
346
|
+
return VariantPairOutcome(
|
|
347
|
+
baseline=outcomes[VariantName.BASELINE],
|
|
348
|
+
candidate=outcomes[VariantName.CANDIDATE],
|
|
349
|
+
schedule=VariantSchedule.SEQUENTIAL,
|
|
350
|
+
skew=None,
|
|
351
|
+
)
|
|
352
|
+
|
|
353
|
+
# -- the start barrier -------------------------------------------------
|
|
354
|
+
|
|
355
|
+
def _require_inputs(
|
|
356
|
+
self,
|
|
357
|
+
pair: VariantPair,
|
|
358
|
+
children: dict[VariantName, EvaluationChild],
|
|
359
|
+
) -> None:
|
|
360
|
+
"""Check every input both variants need, before either child starts."""
|
|
361
|
+
for variant in _VARIANT_ORDER:
|
|
362
|
+
if children[variant].variant is not variant:
|
|
363
|
+
raise ValidationError(
|
|
364
|
+
f"the {variant.value} slot holds a "
|
|
365
|
+
f"{children[variant].variant.value} child",
|
|
366
|
+
code=VARIANT_INPUTS_MISSING,
|
|
367
|
+
details={"variant": variant.value},
|
|
368
|
+
)
|
|
369
|
+
plan = pair.plan(variant)
|
|
370
|
+
for label, path in (
|
|
371
|
+
("compiled config", Path(plan.verifiers_input_config_path)),
|
|
372
|
+
("experiment manifest", Path(plan.experiment_manifest_path)),
|
|
373
|
+
):
|
|
374
|
+
if not path.is_file():
|
|
375
|
+
raise ValidationError(
|
|
376
|
+
f"the {variant.value} variant's {label} is not on disk, "
|
|
377
|
+
"so neither variant may start",
|
|
378
|
+
code=VARIANT_INPUTS_MISSING,
|
|
379
|
+
details={"variant": variant.value, "path": str(path)},
|
|
380
|
+
)
|
|
381
|
+
for skill in plan.skill_paths:
|
|
382
|
+
if not Path(skill).is_dir():
|
|
383
|
+
raise ValidationError(
|
|
384
|
+
f"the {variant.value} variant declares a skill that is "
|
|
385
|
+
"not staged in the run's own input tree",
|
|
386
|
+
code=VARIANT_INPUTS_MISSING,
|
|
387
|
+
details={"variant": variant.value, "path": skill},
|
|
388
|
+
)
|
|
389
|
+
traces = self._traces_path(plan)
|
|
390
|
+
if traces.exists():
|
|
391
|
+
raise ValidationError(
|
|
392
|
+
f"the {variant.value} variant's output directory already "
|
|
393
|
+
"holds evidence; a run writes its own",
|
|
394
|
+
code=VARIANT_INPUTS_MISSING,
|
|
395
|
+
details={"variant": variant.value, "path": str(traces)},
|
|
396
|
+
)
|
|
397
|
+
|
|
398
|
+
# -- starting ----------------------------------------------------------
|
|
399
|
+
|
|
400
|
+
def _start_both(
|
|
401
|
+
self,
|
|
402
|
+
run_id: str,
|
|
403
|
+
*,
|
|
404
|
+
run_root: Path,
|
|
405
|
+
children: dict[VariantName, EvaluationChild],
|
|
406
|
+
) -> LaunchSkew:
|
|
407
|
+
"""Start both children back to back and record how far apart they were."""
|
|
408
|
+
first = self._start_one(run_id, children[VariantName.BASELINE])
|
|
409
|
+
first_monotonic = time.monotonic()
|
|
410
|
+
try:
|
|
411
|
+
second = self._start_one(run_id, children[VariantName.CANDIDATE])
|
|
412
|
+
except BaseException:
|
|
413
|
+
# The pair never existed. Stop the one child that did, so a failed
|
|
414
|
+
# launch does not leave a container talking to a provider.
|
|
415
|
+
self._children.terminate_all(run_id, self._grace)
|
|
416
|
+
raise
|
|
417
|
+
second_monotonic = time.monotonic()
|
|
418
|
+
|
|
419
|
+
skew = LaunchSkew(
|
|
420
|
+
baseline_started_at=first.started_at,
|
|
421
|
+
candidate_started_at=second.started_at,
|
|
422
|
+
seconds=max(second_monotonic - first_monotonic, 0.0),
|
|
423
|
+
first=VariantName.BASELINE,
|
|
424
|
+
)
|
|
425
|
+
write_children_record(
|
|
426
|
+
run_root=run_root,
|
|
427
|
+
run_id=run_id,
|
|
428
|
+
schedule=VariantSchedule.PARALLEL,
|
|
429
|
+
children=[first, second],
|
|
430
|
+
launch_skew_seconds=skew.seconds,
|
|
431
|
+
)
|
|
432
|
+
return skew
|
|
433
|
+
|
|
434
|
+
def _start_one(self, run_id: str, child: EvaluationChild) -> LaunchedChild:
|
|
435
|
+
"""Start one child and register it before anything can fail."""
|
|
436
|
+
try:
|
|
437
|
+
child.start()
|
|
438
|
+
except RunError:
|
|
439
|
+
raise
|
|
440
|
+
except OSError as error:
|
|
441
|
+
raise RunError(
|
|
442
|
+
f"the {child.variant.value} evaluation child could not be "
|
|
443
|
+
f"started: {error.strerror or error}",
|
|
444
|
+
code=VARIANT_CHILD_START_FAILED,
|
|
445
|
+
details={"run_id": run_id, "variant": child.variant.value},
|
|
446
|
+
) from error
|
|
447
|
+
self._children.register(run_id, child)
|
|
448
|
+
return LaunchedChild(
|
|
449
|
+
variant=child.variant,
|
|
450
|
+
pid=child.pid,
|
|
451
|
+
argv_digest=child.argv_digest,
|
|
452
|
+
started_at=self._clock(),
|
|
453
|
+
)
|
|
454
|
+
|
|
455
|
+
# -- watching ----------------------------------------------------------
|
|
456
|
+
|
|
457
|
+
def _watch_both(
|
|
458
|
+
self,
|
|
459
|
+
run_id: str,
|
|
460
|
+
pair: VariantPair,
|
|
461
|
+
children: dict[VariantName, EvaluationChild],
|
|
462
|
+
) -> dict[VariantName, ChildProcessOutcome]:
|
|
463
|
+
"""Poll both children until both have ended, or until one fails."""
|
|
464
|
+
reported: dict[VariantName, VariantProgress | None] = {
|
|
465
|
+
variant: None for variant in _VARIANT_ORDER
|
|
466
|
+
}
|
|
467
|
+
exits: dict[VariantName, int] = {}
|
|
468
|
+
outcomes: dict[VariantName, ChildProcessOutcome] = {}
|
|
469
|
+
|
|
470
|
+
try:
|
|
471
|
+
while len(outcomes) < len(_VARIANT_ORDER):
|
|
472
|
+
raise_if_cancel_requested(self._run_store, run_id)
|
|
473
|
+
for variant in _VARIANT_ORDER:
|
|
474
|
+
if variant in outcomes:
|
|
475
|
+
continue
|
|
476
|
+
child = children[variant]
|
|
477
|
+
code = child.poll()
|
|
478
|
+
progress = self._inspect(pair, variant, code)
|
|
479
|
+
if code is None:
|
|
480
|
+
self._report(run_id, reported, variant, progress)
|
|
481
|
+
continue
|
|
482
|
+
exits[variant] = code
|
|
483
|
+
outcomes[variant] = child.outcome()
|
|
484
|
+
self._children.unregister(run_id, variant)
|
|
485
|
+
self._emit(run_id, VARIANT_COMPLETED, progress)
|
|
486
|
+
reported[variant] = progress
|
|
487
|
+
if code != 0:
|
|
488
|
+
self._stop_sibling(run_id, variant, children, outcomes)
|
|
489
|
+
if len(outcomes) < len(_VARIANT_ORDER):
|
|
490
|
+
time.sleep(self._poll_interval)
|
|
491
|
+
except CancellationError:
|
|
492
|
+
self._children.terminate_all(run_id, self._grace)
|
|
493
|
+
self._collect_remaining(children, outcomes)
|
|
494
|
+
raise
|
|
495
|
+
|
|
496
|
+
self._require_both_succeeded(run_id, exits)
|
|
497
|
+
return outcomes
|
|
498
|
+
|
|
499
|
+
def _watch_one(
|
|
500
|
+
self,
|
|
501
|
+
run_id: str,
|
|
502
|
+
pair: VariantPair,
|
|
503
|
+
child: EvaluationChild,
|
|
504
|
+
) -> ChildProcessOutcome:
|
|
505
|
+
"""Poll one child to its end, reporting phase progress as it goes."""
|
|
506
|
+
variant = child.variant
|
|
507
|
+
last: int | None = None
|
|
508
|
+
try:
|
|
509
|
+
while True:
|
|
510
|
+
raise_if_cancel_requested(self._run_store, run_id)
|
|
511
|
+
code = child.poll()
|
|
512
|
+
progress = self._inspect(pair, variant, code)
|
|
513
|
+
last = self._report_phase_progress(run_id, variant, progress, last)
|
|
514
|
+
if code is not None:
|
|
515
|
+
break
|
|
516
|
+
time.sleep(self._poll_interval)
|
|
517
|
+
except CancellationError:
|
|
518
|
+
self._children.terminate_all(run_id, self._grace)
|
|
519
|
+
raise
|
|
520
|
+
|
|
521
|
+
outcome = child.outcome()
|
|
522
|
+
self._children.unregister(run_id, variant)
|
|
523
|
+
self._require_both_succeeded(run_id, {variant: outcome.exit_code})
|
|
524
|
+
return outcome
|
|
525
|
+
|
|
526
|
+
def _stop_sibling(
|
|
527
|
+
self,
|
|
528
|
+
run_id: str,
|
|
529
|
+
failed: VariantName,
|
|
530
|
+
children: dict[VariantName, EvaluationChild],
|
|
531
|
+
outcomes: dict[VariantName, ChildProcessOutcome],
|
|
532
|
+
) -> None:
|
|
533
|
+
"""Terminate the other side of a pair whose first side failed."""
|
|
534
|
+
for variant in _VARIANT_ORDER:
|
|
535
|
+
if variant is failed or variant in outcomes:
|
|
536
|
+
continue
|
|
537
|
+
sibling = children[variant]
|
|
538
|
+
sibling.terminate(self._grace)
|
|
539
|
+
outcomes[variant] = sibling.outcome()
|
|
540
|
+
self._children.unregister(run_id, variant)
|
|
541
|
+
|
|
542
|
+
def _collect_remaining(
|
|
543
|
+
self,
|
|
544
|
+
children: dict[VariantName, EvaluationChild],
|
|
545
|
+
outcomes: dict[VariantName, ChildProcessOutcome],
|
|
546
|
+
) -> None:
|
|
547
|
+
"""Describe every child that has not been described yet.
|
|
548
|
+
|
|
549
|
+
Called while unwinding, so a child that cannot describe itself is
|
|
550
|
+
skipped rather than allowed to replace the reason the run is stopping.
|
|
551
|
+
"""
|
|
552
|
+
for variant in _VARIANT_ORDER:
|
|
553
|
+
if variant in outcomes:
|
|
554
|
+
continue
|
|
555
|
+
try:
|
|
556
|
+
outcomes[variant] = children[variant].outcome()
|
|
557
|
+
except RunError:
|
|
558
|
+
continue
|
|
559
|
+
|
|
560
|
+
def _require_both_succeeded(
|
|
561
|
+
self, run_id: str, exits: dict[VariantName, int]
|
|
562
|
+
) -> None:
|
|
563
|
+
"""Fail the pair when either side did not finish cleanly."""
|
|
564
|
+
failed = sorted(
|
|
565
|
+
(variant.value, code) for variant, code in exits.items() if code != 0
|
|
566
|
+
)
|
|
567
|
+
if not failed:
|
|
568
|
+
return
|
|
569
|
+
detail: list[JsonValue] = [
|
|
570
|
+
{"variant": variant, "exit_code": code} for variant, code in failed
|
|
571
|
+
]
|
|
572
|
+
names = ", ".join(variant for variant, _ in failed)
|
|
573
|
+
raise RunError(
|
|
574
|
+
f"the {names} evaluation did not finish; a comparison needs both "
|
|
575
|
+
"sides, so the pair failed and the partial evidence was kept",
|
|
576
|
+
code=VARIANT_EXECUTION_FAILED,
|
|
577
|
+
details={"run_id": run_id, "variants": detail},
|
|
578
|
+
)
|
|
579
|
+
|
|
580
|
+
# -- measurement and events -------------------------------------------
|
|
581
|
+
|
|
582
|
+
def _inspect(
|
|
583
|
+
self,
|
|
584
|
+
pair: VariantPair,
|
|
585
|
+
variant: VariantName,
|
|
586
|
+
exit_code: int | None,
|
|
587
|
+
) -> VariantProgress:
|
|
588
|
+
"""Measure one variant from the evidence its child is writing."""
|
|
589
|
+
plan = pair.plan(variant)
|
|
590
|
+
return inspect_progress(
|
|
591
|
+
variant=variant,
|
|
592
|
+
traces_path=self._traces_path(plan),
|
|
593
|
+
total=plan.task_count,
|
|
594
|
+
child_exit_code=exit_code,
|
|
595
|
+
max_concurrent=plan.max_concurrent,
|
|
596
|
+
)
|
|
597
|
+
|
|
598
|
+
def _traces_path(self, plan: VariantExecutionPlan) -> Path:
|
|
599
|
+
"""Where one variant's child appends its finished episodes."""
|
|
600
|
+
return Path(plan.verifiers_output_dir) / TRACES_FILENAME
|
|
601
|
+
|
|
602
|
+
def _announce(self, run_id: str, pair: VariantPair, variant: VariantName) -> None:
|
|
603
|
+
"""Record that one side of the comparison is up."""
|
|
604
|
+
self._emit(
|
|
605
|
+
run_id,
|
|
606
|
+
VARIANT_STARTED,
|
|
607
|
+
pending_progress(variant, pair.plan(variant).task_count),
|
|
608
|
+
)
|
|
609
|
+
|
|
610
|
+
def _report(
|
|
611
|
+
self,
|
|
612
|
+
run_id: str,
|
|
613
|
+
reported: dict[VariantName, VariantProgress | None],
|
|
614
|
+
variant: VariantName,
|
|
615
|
+
progress: VariantProgress,
|
|
616
|
+
) -> None:
|
|
617
|
+
"""Append a progress event only when something actually moved."""
|
|
618
|
+
if reported[variant] == progress:
|
|
619
|
+
return
|
|
620
|
+
self._emit(run_id, VARIANT_PROGRESS, progress)
|
|
621
|
+
reported[variant] = progress
|
|
622
|
+
|
|
623
|
+
def _report_phase_progress(
|
|
624
|
+
self,
|
|
625
|
+
run_id: str,
|
|
626
|
+
variant: VariantName,
|
|
627
|
+
progress: VariantProgress,
|
|
628
|
+
last: int | None,
|
|
629
|
+
) -> int:
|
|
630
|
+
"""Append one sequential phase's progress line when it advances."""
|
|
631
|
+
if progress.completed == last:
|
|
632
|
+
return progress.completed
|
|
633
|
+
with self._cancellation_aware(run_id):
|
|
634
|
+
self._run_store.append(
|
|
635
|
+
run_id,
|
|
636
|
+
phase=None,
|
|
637
|
+
kind=PROGRESS_UPDATED,
|
|
638
|
+
details={
|
|
639
|
+
DETAIL_CURRENT: progress.completed,
|
|
640
|
+
DETAIL_TOTAL: progress.total,
|
|
641
|
+
DETAIL_LABEL: _EPISODE_LABEL.format(variant=variant.value),
|
|
642
|
+
},
|
|
643
|
+
)
|
|
644
|
+
return progress.completed
|
|
645
|
+
|
|
646
|
+
def _emit(self, run_id: str, kind: str, progress: VariantProgress) -> None:
|
|
647
|
+
"""Append one variant event against whatever phase the run is in."""
|
|
648
|
+
with self._cancellation_aware(run_id):
|
|
649
|
+
self._run_store.append(
|
|
650
|
+
run_id,
|
|
651
|
+
phase=None,
|
|
652
|
+
kind=kind,
|
|
653
|
+
details={
|
|
654
|
+
DETAIL_VARIANT: progress.variant,
|
|
655
|
+
DETAIL_COMPLETED: progress.completed,
|
|
656
|
+
DETAIL_TOTAL: progress.total,
|
|
657
|
+
DETAIL_RUNNING: progress.running,
|
|
658
|
+
DETAIL_ERRORED: progress.errored,
|
|
659
|
+
DETAIL_STATE: progress.state,
|
|
660
|
+
},
|
|
661
|
+
)
|
|
662
|
+
|
|
663
|
+
@contextlib.contextmanager
|
|
664
|
+
def _cancellation_aware(self, run_id: str) -> Iterator[None]:
|
|
665
|
+
"""Report a refused append as the cancellation that caused it.
|
|
666
|
+
|
|
667
|
+
A variant event belongs to ``running_variants`` and to no other phase,
|
|
668
|
+
so an append that lands after another process asked the run to stop is
|
|
669
|
+
refused by the run's own state machine. That refusal is not a defect
|
|
670
|
+
and it is not what the caller has to unwind for: between the poller's
|
|
671
|
+
cancellation check and its append, another process moved the run to
|
|
672
|
+
``cancel_requested``, and the cancellation is the fact. Anything else
|
|
673
|
+
that made the append fail is re-raised unchanged.
|
|
674
|
+
"""
|
|
675
|
+
try:
|
|
676
|
+
yield
|
|
677
|
+
except TechtreeError:
|
|
678
|
+
raise_if_cancel_requested(self._run_store, run_id)
|
|
679
|
+
raise
|
|
680
|
+
|
|
681
|
+
|
|
682
|
+
def _utc_now() -> datetime:
|
|
683
|
+
"""Return the current instant in UTC."""
|
|
684
|
+
return datetime.now(UTC)
|