techtree 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- techtree/__init__.py +35 -0
- techtree/__main__.py +14 -0
- techtree/canonical.py +239 -0
- techtree/catalog/__init__.py +25 -0
- techtree/catalog/repository.py +400 -0
- techtree/catalog/service.py +419 -0
- techtree/cli/__init__.py +1 -0
- techtree/cli/app.py +416 -0
- techtree/cli/commands/__init__.py +1 -0
- techtree/cli/commands/climb.py +1223 -0
- techtree/cli/commands/doctor.py +147 -0
- techtree/cli/commands/engine.py +207 -0
- techtree/cli/commands/proof.py +556 -0
- techtree/cli/commands/publish.py +447 -0
- techtree/cli/commands/release.py +303 -0
- techtree/cli/commands/run.py +1067 -0
- techtree/cli/commands/setup.py +181 -0
- techtree/cli/commands/skill.py +221 -0
- techtree/cli/commands/uplift.py +698 -0
- techtree/cli/commands/withdraw.py +212 -0
- techtree/cli/confirm.py +47 -0
- techtree/cli/context.py +96 -0
- techtree/cli/invoke.py +220 -0
- techtree/cli/output.py +280 -0
- techtree/constants.py +138 -0
- techtree/crypto.py +128 -0
- techtree/doctor/__init__.py +1 -0
- techtree/doctor/checks.py +675 -0
- techtree/doctor/execution_checks.py +435 -0
- techtree/doctor/service.py +326 -0
- techtree/drafts/__init__.py +32 -0
- techtree/drafts/source.py +146 -0
- techtree/drafts/store.py +992 -0
- techtree/engines/__init__.py +1 -0
- techtree/engines/bundle.py +251 -0
- techtree/engines/installer.py +679 -0
- techtree/engines/registry.py +235 -0
- techtree/engines/runner.py +170 -0
- techtree/errors.py +262 -0
- techtree/fs.py +234 -0
- techtree/harness.py +108 -0
- techtree/identity/__init__.py +41 -0
- techtree/identity/models.py +113 -0
- techtree/identity/service.py +199 -0
- techtree/identity/store.py +263 -0
- techtree/ids.py +85 -0
- techtree/manifests/__init__.py +39 -0
- techtree/manifests/builder.py +433 -0
- techtree/manifests/compare.py +376 -0
- techtree/models/__init__.py +282 -0
- techtree/models/base.py +201 -0
- techtree/models/campaign.py +484 -0
- techtree/models/catalog.py +227 -0
- techtree/models/cli.py +151 -0
- techtree/models/climb.py +254 -0
- techtree/models/data_policy.py +130 -0
- techtree/models/engine.py +156 -0
- techtree/models/episode_receipt.py +130 -0
- techtree/models/evaluation_backend.py +113 -0
- techtree/models/experiment.py +154 -0
- techtree/models/run.py +214 -0
- techtree/models/skill.py +156 -0
- techtree/models/uplift_report.py +158 -0
- techtree/models/validation.py +299 -0
- techtree/paths.py +116 -0
- techtree/presentation/__init__.py +31 -0
- techtree/presentation/build.py +1242 -0
- techtree/presentation/compact.py +246 -0
- techtree/presentation/evidence.py +169 -0
- techtree/presentation/models.py +358 -0
- techtree/presentation/rich.py +312 -0
- techtree/presentation/sanitize.py +156 -0
- techtree/publication/__init__.py +44 -0
- techtree/publication/address.py +180 -0
- techtree/publication/coordinates.py +26 -0
- techtree/publication/journal.py +212 -0
- techtree/publication/keccak.py +183 -0
- techtree/publication/models.py +209 -0
- techtree/publication/offer.py +35 -0
- techtree/publication/service.py +618 -0
- techtree/publication/transport.py +296 -0
- techtree/publication/verify.py +242 -0
- techtree/publication/withdraw.py +156 -0
- techtree/py.typed +0 -0
- techtree/receipts/__init__.py +52 -0
- techtree/receipts/bundle.py +578 -0
- techtree/receipts/compare.py +1065 -0
- techtree/receipts/episode.py +672 -0
- techtree/receipts/execution.py +630 -0
- techtree/receipts/observed.py +474 -0
- techtree/receipts/set.py +336 -0
- techtree/receipts/uplift.py +655 -0
- techtree/receipts/verify.py +1055 -0
- techtree/release/__init__.py +9 -0
- techtree/release/bootstrap.py +509 -0
- techtree/release/checks.py +376 -0
- techtree/release/document.py +125 -0
- techtree/release/generate.py +221 -0
- techtree/release/models.py +293 -0
- techtree/release/provenance.py +109 -0
- techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
- techtree/resources/catalog/catalog.json +32 -0
- techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
- techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
- techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
- techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
- techtree/resources/engines/default/engine.json +20 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
- techtree/resources/engines/default/pyproject.toml +23 -0
- techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
- techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
- techtree/resources/engines/default/tools/normalize_validation.py +222 -0
- techtree/resources/engines/default/uv.lock +1758 -0
- techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
- techtree/resources/release/build-provenance.json +4 -0
- techtree/resources/release/release-core.json +24 -0
- techtree/runs/__init__.py +31 -0
- techtree/runs/artifacts.py +750 -0
- techtree/runs/child_registry.py +228 -0
- techtree/runs/events.py +478 -0
- techtree/runs/executor.py +140 -0
- techtree/runs/fake.py +741 -0
- techtree/runs/launcher.py +253 -0
- techtree/runs/machine.py +489 -0
- techtree/runs/real.py +789 -0
- techtree/runs/service.py +616 -0
- techtree/runs/store.py +555 -0
- techtree/runs/validation.py +259 -0
- techtree/runs/variants.py +684 -0
- techtree/settings.py +143 -0
- techtree/skills/__init__.py +14 -0
- techtree/skills/archive.py +282 -0
- techtree/skills/policy.py +62 -0
- techtree/skills/scanner.py +394 -0
- techtree/skills/service.py +752 -0
- techtree/skills/starter.py +434 -0
- techtree/tasksets/__init__.py +1 -0
- techtree/tasksets/membership.py +269 -0
- techtree/tasksets/provider.py +207 -0
- techtree/tasksets/resolver.py +311 -0
- techtree/tasksets/service.py +484 -0
- techtree/tasksets/verifiers_cli.py +538 -0
- techtree/uplift/__init__.py +20 -0
- techtree/uplift/context.py +544 -0
- techtree/uplift/derive.py +203 -0
- techtree/uplift/public_tasks.py +151 -0
- techtree/uplift/service.py +719 -0
- techtree/uplift/source.py +160 -0
- techtree/verifiers/__init__.py +31 -0
- techtree/verifiers/budget.py +219 -0
- techtree/verifiers/child.py +633 -0
- techtree/verifiers/compiler.py +432 -0
- techtree/verifiers/config.py +365 -0
- techtree/verifiers/credentials.py +321 -0
- techtree/verifiers/image.py +126 -0
- techtree/verifiers/models.py +527 -0
- techtree/verifiers/outputs.py +368 -0
- techtree/verifiers/progress.py +192 -0
- techtree/verifiers/supervisor.py +341 -0
- techtree/verifiers/verify.py +782 -0
- techtree/version.py +39 -0
- techtree/worker/__init__.py +18 -0
- techtree/worker/execute.py +487 -0
- techtree/worker/main.py +57 -0
- techtree-0.1.0.dist-info/METADATA +344 -0
- techtree-0.1.0.dist-info/RECORD +174 -0
- techtree-0.1.0.dist-info/WHEEL +4 -0
- techtree-0.1.0.dist-info/entry_points.txt +3 -0
- techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,365 @@
|
|
|
1
|
+
"""The only Verifiers configuration Techtree may emit. Spec section 6.7.
|
|
2
|
+
|
|
3
|
+
This module is an allow-list, and the allow-list is the control. Verifiers has
|
|
4
|
+
a large and growing set of knobs; a Campaign that could reach any of them could
|
|
5
|
+
introduce a second difference between baseline and candidate without anyone
|
|
6
|
+
being able to name it afterwards. Every model here forbids extra keys, so a
|
|
7
|
+
knob Techtree has not deliberately modelled is unrepresentable rather than
|
|
8
|
+
merely discouraged.
|
|
9
|
+
|
|
10
|
+
Several fields are typed as literals rather than validated at runtime, because
|
|
11
|
+
"the value is wrong" and "the value cannot be spelled" are different guarantees:
|
|
12
|
+
|
|
13
|
+
``push: Literal[False]``
|
|
14
|
+
``EvalConfig.push`` defaults to **true** upstream and uploads the complete
|
|
15
|
+
Episode — prompts and subject replies included — to the Prime platform
|
|
16
|
+
(``docs/verifiers-eval.md``, finding E1). A config that merely forgets to
|
|
17
|
+
set it exfiltrates the participant's trajectories.
|
|
18
|
+
``rich: Literal[None]``
|
|
19
|
+
The dashboard is the whole output when it is on; a captured child needs log
|
|
20
|
+
lines. Upstream's ``rich`` is a table, not a flag, and **null is the only
|
|
21
|
+
spelling that turns the dashboard off**. Measured against the pinned build:
|
|
22
|
+
an omitted key resolves to ``{"show_logs": false}``, which is the dashboard
|
|
23
|
+
on *and* the log lines suppressed — the exact failure this lock-down
|
|
24
|
+
exists to prevent. So the danger here is a key that is missing, not a key
|
|
25
|
+
set to true, and :func:`emitted_document` writes this one null explicitly
|
|
26
|
+
while every other unset optional stays absent.
|
|
27
|
+
``shuffle: Literal[False]`` and ``num_rollouts: Literal[1]``
|
|
28
|
+
Decisions document 0001. There is no seed anywhere in the protocol, so a
|
|
29
|
+
shuffled run could not be reproduced by anyone.
|
|
30
|
+
``use_bundled_skill: Literal[False]``
|
|
31
|
+
A bundled skill catalogue is an uncontrolled second difference.
|
|
32
|
+
|
|
33
|
+
``disabled_tools`` is absent by construction. The native Hermes harness accepts
|
|
34
|
+
it at config time and then refuses it in the middle of a run, after Docker has
|
|
35
|
+
been provisioned (``docs/verifiers-eval.md``, finding E3), so the only safe
|
|
36
|
+
place to reject it is here.
|
|
37
|
+
|
|
38
|
+
The Docker table carries ``allow`` and ``block``, which spec section 6.7 does
|
|
39
|
+
not model. Upstream's default egress policy is unrestricted, so without them a
|
|
40
|
+
Campaign declaring ``network_policy: "restricted"`` would compile to a
|
|
41
|
+
container with open network access.
|
|
42
|
+
|
|
43
|
+
The emitted document is **JSON**, and that is a safety property rather than a
|
|
44
|
+
taste. TOML has no null literal, so a TOML document cannot say "no dashboard"
|
|
45
|
+
at all — it can only leave ``rich`` out, which is the dangerous value. The
|
|
46
|
+
engine reads either format and picks its parser from the file extension, so
|
|
47
|
+
the configuration Techtree writes is named ``.json`` and every locked-down
|
|
48
|
+
setting stays expressible only as its safe value, in the file, where the run's
|
|
49
|
+
own inputs record it.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
from __future__ import annotations
|
|
53
|
+
|
|
54
|
+
import json
|
|
55
|
+
import re
|
|
56
|
+
from typing import Any, Final, Literal, Self
|
|
57
|
+
|
|
58
|
+
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
|
59
|
+
|
|
60
|
+
from techtree.errors import ValidationError
|
|
61
|
+
from techtree.models.campaign import CREDENTIAL_ENV_PATTERN
|
|
62
|
+
|
|
63
|
+
__all__ = [
|
|
64
|
+
"ALLOWED_CLIENT_HEADERS",
|
|
65
|
+
"EVAL_CONFIG_INVALID",
|
|
66
|
+
"IMAGE_DIGEST_PATTERN",
|
|
67
|
+
"OPEN_BLOCK",
|
|
68
|
+
"OPEN_NETWORK",
|
|
69
|
+
"RESTRICTED_BLOCK",
|
|
70
|
+
"RESTRICTED_NETWORK",
|
|
71
|
+
"DockerRuntimeToml",
|
|
72
|
+
"EnvToml",
|
|
73
|
+
"EvalClientToml",
|
|
74
|
+
"EvalToml",
|
|
75
|
+
"HermesHarnessToml",
|
|
76
|
+
"SamplingToml",
|
|
77
|
+
"SubjectAgentToml",
|
|
78
|
+
"TasksetToml",
|
|
79
|
+
"TimeoutToml",
|
|
80
|
+
"config_to_json_bytes",
|
|
81
|
+
"egress_for",
|
|
82
|
+
"emitted_document",
|
|
83
|
+
]
|
|
84
|
+
|
|
85
|
+
#: Stable error code. Spec section 6.7.
|
|
86
|
+
EVAL_CONFIG_INVALID: Final = "eval_config_invalid"
|
|
87
|
+
|
|
88
|
+
#: Headers Techtree is allowed to declare on the evaluation client. Empty: the
|
|
89
|
+
#: only supported profile routes through the pinned client's own resolution,
|
|
90
|
+
#: and a header Techtree invents is a routing decision nobody reviewed. The
|
|
91
|
+
#: engine may still *add* headers of its own when it resolves the config
|
|
92
|
+
#: (``docs/verifiers-eval.md``); that is upstream's business, not Techtree's.
|
|
93
|
+
ALLOWED_CLIENT_HEADERS: Final[frozenset[str]] = frozenset()
|
|
94
|
+
|
|
95
|
+
#: Framework-only egress: the interception endpoint and nothing else, which is
|
|
96
|
+
#: what a restricted subject runtime means. Upstream normalizes an empty
|
|
97
|
+
#: allow-list to exactly this pair, so Techtree declares the normalized form
|
|
98
|
+
#: rather than the shorthand — otherwise the configuration Techtree compiled
|
|
99
|
+
#: and the configuration the engine resolved would disagree at ``block`` on
|
|
100
|
+
#: every restricted run.
|
|
101
|
+
RESTRICTED_NETWORK: Final[tuple[str, ...]] = ()
|
|
102
|
+
RESTRICTED_BLOCK: Final[tuple[str, ...]] = ("*",)
|
|
103
|
+
#: Unrestricted egress, upstream's own default spelling.
|
|
104
|
+
OPEN_NETWORK: Final[tuple[str, ...]] = ("*",)
|
|
105
|
+
OPEN_BLOCK: Final[tuple[str, ...]] = ()
|
|
106
|
+
|
|
107
|
+
#: An OCI reference that names content rather than a moving tag.
|
|
108
|
+
IMAGE_DIGEST_PATTERN: Final = r"@sha256:[0-9a-f]{64}$"
|
|
109
|
+
|
|
110
|
+
_CREDENTIAL_ENV_RE: Final = re.compile(CREDENTIAL_ENV_PATTERN)
|
|
111
|
+
_IMAGE_DIGEST_RE: Final = re.compile(IMAGE_DIGEST_PATTERN)
|
|
112
|
+
|
|
113
|
+
#: The settings whose null the emitted document must state out loud, because
|
|
114
|
+
#: leaving the key out would select something else. Only ``rich`` qualifies:
|
|
115
|
+
#: every other optional in this module defaults to the same ``None`` upstream.
|
|
116
|
+
_NULL_IS_THE_DECISION: Final[frozenset[str]] = frozenset({"rich"})
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class TomlModel(BaseModel):
|
|
120
|
+
"""A frozen, extra-forbidden fragment of the emitted configuration."""
|
|
121
|
+
|
|
122
|
+
model_config = ConfigDict(frozen=True, extra="forbid", validate_default=True)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
class EvalClientToml(TomlModel):
|
|
126
|
+
"""The OpenAI-compatible endpoint the evaluation runs against.
|
|
127
|
+
|
|
128
|
+
``base_url`` is deliberately omitted for the supported profile. The pinned
|
|
129
|
+
client resolves a ``PRIME_API_KEY``-keyed endpoint from the environment and
|
|
130
|
+
the active Prime CLI configuration, and writing a URL here would freeze a
|
|
131
|
+
deployment detail into a run's inputs.
|
|
132
|
+
"""
|
|
133
|
+
|
|
134
|
+
type: Literal["eval"] = "eval"
|
|
135
|
+
api_key_var: str
|
|
136
|
+
base_url: str | None = None
|
|
137
|
+
headers: dict[str, str] = Field(default_factory=dict)
|
|
138
|
+
|
|
139
|
+
@model_validator(mode="after")
|
|
140
|
+
def _check_the_client_names_a_variable_and_no_headers(self) -> Self:
|
|
141
|
+
"""Reject a credential value, and any header Techtree may not declare."""
|
|
142
|
+
if _CREDENTIAL_ENV_RE.fullmatch(self.api_key_var) is None:
|
|
143
|
+
raise ValueError(
|
|
144
|
+
"api_key_var must be an uppercase environment-variable name, "
|
|
145
|
+
"never a credential value"
|
|
146
|
+
)
|
|
147
|
+
unknown = sorted(set(self.headers) - ALLOWED_CLIENT_HEADERS)
|
|
148
|
+
if unknown:
|
|
149
|
+
raise ValueError(f"Techtree does not declare client headers; got {unknown}")
|
|
150
|
+
return self
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
class SamplingToml(TomlModel):
|
|
154
|
+
"""How the subject model is sampled."""
|
|
155
|
+
|
|
156
|
+
temperature: float = Field(ge=0.0, le=2.0)
|
|
157
|
+
max_tokens: int = Field(ge=1)
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
class HermesHarnessToml(TomlModel):
|
|
161
|
+
"""The pinned Hermes Agent harness and the skills inserted into it."""
|
|
162
|
+
|
|
163
|
+
id: Literal["hermes-agent"] = "hermes-agent"
|
|
164
|
+
version: str = Field(min_length=1)
|
|
165
|
+
use_bundled_skill: Literal[False] = False
|
|
166
|
+
skills: list[str] = Field(default_factory=list)
|
|
167
|
+
|
|
168
|
+
@model_validator(mode="after")
|
|
169
|
+
def _check_skill_paths_are_absolute_and_distinct(self) -> Self:
|
|
170
|
+
"""Reject a relative or repeated skill path."""
|
|
171
|
+
for path in self.skills:
|
|
172
|
+
if not path.startswith("/"):
|
|
173
|
+
raise ValueError(
|
|
174
|
+
f"skill paths are absolute run-owned paths; got {path!r}"
|
|
175
|
+
)
|
|
176
|
+
if len(set(self.skills)) != len(self.skills):
|
|
177
|
+
raise ValueError("a harness mounts each skill exactly once")
|
|
178
|
+
return self
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
class DockerRuntimeToml(TomlModel):
|
|
182
|
+
"""Where the subject agent executes."""
|
|
183
|
+
|
|
184
|
+
type: Literal["docker"] = "docker"
|
|
185
|
+
image: str = Field(min_length=1)
|
|
186
|
+
allow: list[str] = Field(default_factory=list)
|
|
187
|
+
block: list[str] = Field(default_factory=list)
|
|
188
|
+
cpu: float | None = Field(default=None, gt=0.0)
|
|
189
|
+
memory: float | None = Field(default=None, gt=0.0)
|
|
190
|
+
|
|
191
|
+
@property
|
|
192
|
+
def image_is_digest_pinned(self) -> bool:
|
|
193
|
+
"""Whether the image reference names content rather than a tag."""
|
|
194
|
+
return _IMAGE_DIGEST_RE.search(self.image) is not None
|
|
195
|
+
|
|
196
|
+
@property
|
|
197
|
+
def network_is_restricted(self) -> bool:
|
|
198
|
+
"""Whether egress is anything narrower than unrestricted."""
|
|
199
|
+
return list(self.allow) != list(OPEN_NETWORK) or bool(self.block)
|
|
200
|
+
|
|
201
|
+
@model_validator(mode="after")
|
|
202
|
+
def _check_the_egress_lists_are_not_both_concrete(self) -> Self:
|
|
203
|
+
"""Reject the one egress combination upstream refuses outright."""
|
|
204
|
+
if self.allow and list(self.allow) != list(OPEN_NETWORK) and self.block:
|
|
205
|
+
raise ValueError(
|
|
206
|
+
"a concrete allow list and a block list are mutually exclusive"
|
|
207
|
+
)
|
|
208
|
+
return self
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
class TimeoutToml(TomlModel):
|
|
212
|
+
"""Per-rollout Verifiers lifecycle limits.
|
|
213
|
+
|
|
214
|
+
Every one of these defaults to ``None`` upstream, and ``None`` means "no
|
|
215
|
+
limit": the pinned build wraps each phase in ``asyncio.timeout(value)``, and
|
|
216
|
+
``asyncio.timeout(None)`` is a no-op. A Campaign that declares
|
|
217
|
+
``timeout_seconds`` and cannot reach this table has declared nothing, which
|
|
218
|
+
is the whole reason the table exists here (decisions document 0029, layer
|
|
219
|
+
A).
|
|
220
|
+
"""
|
|
221
|
+
|
|
222
|
+
setup: float | None = Field(default=None, gt=0.0)
|
|
223
|
+
rollout: float | None = Field(default=None, gt=0.0)
|
|
224
|
+
finalize: float | None = Field(default=None, gt=0.0)
|
|
225
|
+
scoring: float | None = Field(default=None, gt=0.0)
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
class SubjectAgentToml(TomlModel):
|
|
229
|
+
"""The one seat v0.1 evaluates, named for the role it plays."""
|
|
230
|
+
|
|
231
|
+
harness: HermesHarnessToml
|
|
232
|
+
runtime: DockerRuntimeToml
|
|
233
|
+
max_turns: int | None = Field(default=None, ge=1)
|
|
234
|
+
max_input_tokens: int | None = Field(default=None, ge=1)
|
|
235
|
+
max_output_tokens: int | None = Field(default=None, ge=1)
|
|
236
|
+
max_total_tokens: int | None = Field(default=None, ge=1)
|
|
237
|
+
timeout: TimeoutToml = Field(default_factory=TimeoutToml)
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
class TasksetToml(TomlModel):
|
|
241
|
+
"""Which taskset the environment seeds from.
|
|
242
|
+
|
|
243
|
+
Only the identifier. Every other taskset field on the pinned reference
|
|
244
|
+
Taskset is a default Techtree does not vary, and a field Techtree does not
|
|
245
|
+
vary should not appear in a document a reader has to check.
|
|
246
|
+
"""
|
|
247
|
+
|
|
248
|
+
id: str = Field(min_length=1)
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
class EnvToml(TomlModel):
|
|
252
|
+
"""The environment block: the taskset, the subject seat, and the bound.
|
|
253
|
+
|
|
254
|
+
The seat is spelled ``subject`` because the reference package's ``Env``
|
|
255
|
+
declares a field of that name; Verifiers stamps the field name onto every
|
|
256
|
+
trace as ``agent.name`` (spec section 6.5). Against an environment without
|
|
257
|
+
that seat the whole configuration is rejected at parse time
|
|
258
|
+
(``docs/verifiers-eval.md``, finding E0).
|
|
259
|
+
"""
|
|
260
|
+
|
|
261
|
+
taskset: TasksetToml
|
|
262
|
+
subject: SubjectAgentToml
|
|
263
|
+
max_concurrent_agents: int = Field(default=1, ge=1)
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
class EvalToml(TomlModel):
|
|
267
|
+
"""One resolved Techtree experiment as a Verifiers evaluation."""
|
|
268
|
+
|
|
269
|
+
model: str = Field(min_length=1)
|
|
270
|
+
client: EvalClientToml
|
|
271
|
+
sampling: SamplingToml
|
|
272
|
+
env: EnvToml
|
|
273
|
+
num_tasks: int = Field(ge=1)
|
|
274
|
+
num_rollouts: Literal[1] = 1
|
|
275
|
+
shuffle: Literal[False] = False
|
|
276
|
+
max_concurrent: int = Field(ge=1)
|
|
277
|
+
rich: Literal[None] = None
|
|
278
|
+
push: Literal[False] = False
|
|
279
|
+
output_dir: str = Field(min_length=1)
|
|
280
|
+
|
|
281
|
+
@model_validator(mode="after")
|
|
282
|
+
def _check_the_run_is_bounded_and_run_owned(self) -> Self:
|
|
283
|
+
"""Reject a relative output directory or an unbounded episode fan-out."""
|
|
284
|
+
if not self.output_dir.startswith("/"):
|
|
285
|
+
raise ValueError(
|
|
286
|
+
"output_dir is an absolute run-owned path; a relative path "
|
|
287
|
+
"lands wherever the child happened to be started"
|
|
288
|
+
)
|
|
289
|
+
if self.env.max_concurrent_agents > self.max_concurrent:
|
|
290
|
+
raise ValueError(
|
|
291
|
+
"max_concurrent_agents cannot exceed max_concurrent; the "
|
|
292
|
+
"product is the number of live subject runs"
|
|
293
|
+
)
|
|
294
|
+
return self
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def egress_for(network_policy: str) -> tuple[list[str], list[str]]:
|
|
298
|
+
"""Return the ``(allow, block)`` pair one Campaign network policy compiles to.
|
|
299
|
+
|
|
300
|
+
The Campaign speaks in intent — restricted or open — and upstream speaks in
|
|
301
|
+
two lists whose meaning depends on each other. This is the single place the
|
|
302
|
+
two vocabularies meet, and it emits the already-normalized form so nothing
|
|
303
|
+
downstream has to know that upstream would have rewritten the shorthand.
|
|
304
|
+
"""
|
|
305
|
+
if network_policy == "restricted":
|
|
306
|
+
return list(RESTRICTED_NETWORK), list(RESTRICTED_BLOCK)
|
|
307
|
+
if network_policy == "open":
|
|
308
|
+
return list(OPEN_NETWORK), list(OPEN_BLOCK)
|
|
309
|
+
raise ValidationError(
|
|
310
|
+
f"{network_policy!r} is not a network policy Techtree can compile",
|
|
311
|
+
code=EVAL_CONFIG_INVALID,
|
|
312
|
+
details={"network_policy": network_policy},
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
def emitted_document(config: EvalToml) -> dict[str, Any]:
|
|
317
|
+
"""Return exactly the mapping Techtree writes for one configuration.
|
|
318
|
+
|
|
319
|
+
``exclude_none`` is kept for every field but one, because for every other
|
|
320
|
+
optional here a null and an absence mean the same thing to the engine and
|
|
321
|
+
an absence says it more honestly. ``client.base_url`` is left out so the
|
|
322
|
+
pinned client resolves the endpoint itself; an unset token ceiling or
|
|
323
|
+
timeout phase is a limit the Campaign did not declare, and upstream's own
|
|
324
|
+
default for each is the same ``None`` it would have been written as. In
|
|
325
|
+
every one of those cases the document should say only what Techtree
|
|
326
|
+
actually decided.
|
|
327
|
+
|
|
328
|
+
``rich`` is the exception, and it is why the omission is not simply
|
|
329
|
+
switched off wholesale: its null is the decision, and dropping the key
|
|
330
|
+
selects the opposite (see the module docstring). So it is kept, which is
|
|
331
|
+
safe precisely because :class:`EvalToml` can spell no other value for it.
|
|
332
|
+
"""
|
|
333
|
+
document: dict[str, Any] = config.model_dump(mode="json")
|
|
334
|
+
return {
|
|
335
|
+
key: _without_nulls(value)
|
|
336
|
+
for key, value in document.items()
|
|
337
|
+
if value is not None or key in _NULL_IS_THE_DECISION
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def _without_nulls(value: Any) -> Any:
|
|
342
|
+
"""Drop every unset optional from one nested value."""
|
|
343
|
+
if isinstance(value, dict):
|
|
344
|
+
return {
|
|
345
|
+
key: _without_nulls(inner)
|
|
346
|
+
for key, inner in value.items()
|
|
347
|
+
if inner is not None
|
|
348
|
+
}
|
|
349
|
+
return value
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
def config_to_json_bytes(config: EvalToml) -> bytes:
|
|
353
|
+
"""Serialize one configuration to deterministic JSON bytes.
|
|
354
|
+
|
|
355
|
+
Key order is declaration order, so the same configuration always produces
|
|
356
|
+
the same bytes. The engine chooses its parser from the file's extension,
|
|
357
|
+
so these bytes belong in a file named ``.json``.
|
|
358
|
+
"""
|
|
359
|
+
try:
|
|
360
|
+
return json.dumps(emitted_document(config), indent=2).encode("utf-8") + b"\n"
|
|
361
|
+
except (TypeError, ValueError) as error:
|
|
362
|
+
raise ValidationError(
|
|
363
|
+
f"the compiled evaluation config is not serializable: {error}",
|
|
364
|
+
code=EVAL_CONFIG_INVALID,
|
|
365
|
+
) from error
|
|
@@ -0,0 +1,321 @@
|
|
|
1
|
+
"""Whether the evaluation endpoint can authenticate. Spec section 6.9.
|
|
2
|
+
|
|
3
|
+
Two rules shape everything in this module.
|
|
4
|
+
|
|
5
|
+
*A secret is checked, never carried.* No function here returns a credential
|
|
6
|
+
value, writes one, or puts one in an error message, a detail dictionary, or a
|
|
7
|
+
log line. The only object that ever holds the value is the child process
|
|
8
|
+
environment :func:`scrubbed_child_environment` builds, and that dictionary is
|
|
9
|
+
handed straight to the child. :func:`redacted_environment` exists so that the
|
|
10
|
+
same dictionary can be described in a diagnostic without being disclosed.
|
|
11
|
+
|
|
12
|
+
*Evaluation auth is not host auth.* The credential this module diagnoses buys
|
|
13
|
+
model tokens for the evaluated subject. It is unrelated to whatever the
|
|
14
|
+
operator's own Hermes is authenticated with, and confusing the two produces the
|
|
15
|
+
worst possible failure: a run that looks configured, provisions Docker, and
|
|
16
|
+
then discovers thirty seconds later that nothing can answer. Spec sections 6.9
|
|
17
|
+
and 6.18 keep them separate, and so does the wording of every message here.
|
|
18
|
+
|
|
19
|
+
The pinned client resolves a ``PRIME_API_KEY``-named credential from the
|
|
20
|
+
environment first and from the active Prime CLI configuration second, returning
|
|
21
|
+
the literal string ``"EMPTY"`` when it finds neither
|
|
22
|
+
(``docs/verifiers-eval.md``). A missing credential therefore does not fail at
|
|
23
|
+
startup; it fails at the first model call, after the container is up. That is
|
|
24
|
+
the whole reason this check runs before a child is launched.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import json
|
|
30
|
+
import os
|
|
31
|
+
from collections.abc import Iterable, Mapping
|
|
32
|
+
from pathlib import Path
|
|
33
|
+
from typing import Final, Literal
|
|
34
|
+
|
|
35
|
+
from techtree.errors import AuthenticationError
|
|
36
|
+
from techtree.models.base import NonEmptyString, ProtocolModel
|
|
37
|
+
from techtree.models.campaign import ModelSpec
|
|
38
|
+
from techtree.models.cli import NextAction
|
|
39
|
+
from techtree.models.engine import EngineInstallation
|
|
40
|
+
|
|
41
|
+
__all__ = [
|
|
42
|
+
"MODEL_CREDENTIALS_MISSING",
|
|
43
|
+
"PRIME_CONFIG_RELATIVE_PATH",
|
|
44
|
+
"PRIME_CREDENTIAL_ENV",
|
|
45
|
+
"PRIME_ENVIRONMENT",
|
|
46
|
+
"CredentialStatus",
|
|
47
|
+
"credential_status",
|
|
48
|
+
"redacted_environment",
|
|
49
|
+
"require_credentials",
|
|
50
|
+
"scrubbed_child_environment",
|
|
51
|
+
]
|
|
52
|
+
|
|
53
|
+
#: Stable error code. Spec section 6.9.
|
|
54
|
+
MODEL_CREDENTIALS_MISSING: Final = "model_credentials_missing"
|
|
55
|
+
|
|
56
|
+
#: The credential name the pinned client gives its own resolution to. Any other
|
|
57
|
+
#: name is read from the environment and nowhere else.
|
|
58
|
+
PRIME_CREDENTIAL_ENV: Final = "PRIME_API_KEY"
|
|
59
|
+
|
|
60
|
+
#: Where the Prime CLI keeps its configuration, relative to ``HOME``.
|
|
61
|
+
PRIME_CONFIG_RELATIVE_PATH: Final = (".prime", "config.json")
|
|
62
|
+
|
|
63
|
+
#: Prime variables the pinned client reads when it resolves an endpoint. The
|
|
64
|
+
#: key itself is not here: it is forwarded by name from the Campaign.
|
|
65
|
+
PRIME_ENVIRONMENT: Final[tuple[str, ...]] = (
|
|
66
|
+
"PRIME_INFERENCE_URL",
|
|
67
|
+
"PRIME_TEAM_ID",
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
#: The only host variables a child inherits. ``PATH`` so ordinary system tools
|
|
71
|
+
#: and the container runtime resolve, ``HOME`` because the Prime CLI
|
|
72
|
+
#: configuration and package caches hang off it, ``TMPDIR`` so scratch files
|
|
73
|
+
#: land where the host expects them.
|
|
74
|
+
_BASE_ENVIRONMENT: Final[tuple[str, ...]] = ("PATH", "HOME", "TMPDIR")
|
|
75
|
+
|
|
76
|
+
_PRIME_CONFIG_KEY: Final = "api_key"
|
|
77
|
+
|
|
78
|
+
#: What the Prime CLI configuration under one ``HOME`` supplies.
|
|
79
|
+
type _PrimeConfigState = Literal["usable", "signed_out", "malformed", "absent"]
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class CredentialStatus(ProtocolModel):
|
|
83
|
+
"""Whether one model endpoint can authenticate, and from where.
|
|
84
|
+
|
|
85
|
+
``detail`` is operator-facing prose. It names the variable and the place it
|
|
86
|
+
was looked for, never a value or a fragment of one.
|
|
87
|
+
|
|
88
|
+
``malformed_prime_config`` is its own answer rather than another way of
|
|
89
|
+
saying "missing": a configuration file that cannot be read is a broken
|
|
90
|
+
store, and telling somebody who has signed in that they have not is the
|
|
91
|
+
kind of advice that sends them round the loop again.
|
|
92
|
+
"""
|
|
93
|
+
|
|
94
|
+
provider: NonEmptyString
|
|
95
|
+
credential_env: NonEmptyString
|
|
96
|
+
available: bool
|
|
97
|
+
source: Literal["environment", "prime_config", "missing", "malformed_prime_config"]
|
|
98
|
+
detail: NonEmptyString
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def credential_status(
|
|
102
|
+
model: ModelSpec, *, environ: Mapping[str, str] | None = None
|
|
103
|
+
) -> CredentialStatus:
|
|
104
|
+
"""Report whether the declared credential can be resolved.
|
|
105
|
+
|
|
106
|
+
Presence only. The value is never read into a return, a log, or an error.
|
|
107
|
+
|
|
108
|
+
``environ`` is the environment the question is asked *about*, which is not
|
|
109
|
+
always the one this process happens to have. A readiness check runs in an
|
|
110
|
+
operator's terminal but has to answer for the environment a detached run
|
|
111
|
+
would get, and a check that answered for its own terminal instead would be
|
|
112
|
+
able to say "ready" about a run that cannot authenticate.
|
|
113
|
+
"""
|
|
114
|
+
source = os.environ if environ is None else environ
|
|
115
|
+
name = model.credential_env
|
|
116
|
+
if source.get(name):
|
|
117
|
+
return CredentialStatus(
|
|
118
|
+
provider=model.provider,
|
|
119
|
+
credential_env=name,
|
|
120
|
+
available=True,
|
|
121
|
+
source="environment",
|
|
122
|
+
detail=f"{name} is set in this environment.",
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
state = (
|
|
126
|
+
_prime_config_state(source.get("HOME"))
|
|
127
|
+
if name == PRIME_CREDENTIAL_ENV
|
|
128
|
+
else "absent"
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
if state == "usable":
|
|
132
|
+
return CredentialStatus(
|
|
133
|
+
provider=model.provider,
|
|
134
|
+
credential_env=name,
|
|
135
|
+
available=True,
|
|
136
|
+
source="prime_config",
|
|
137
|
+
detail=(
|
|
138
|
+
"the active Prime CLI configuration holds a key the pinned "
|
|
139
|
+
"evaluation client can use."
|
|
140
|
+
),
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
if state == "malformed":
|
|
144
|
+
return CredentialStatus(
|
|
145
|
+
provider=model.provider,
|
|
146
|
+
credential_env=name,
|
|
147
|
+
available=False,
|
|
148
|
+
source="malformed_prime_config",
|
|
149
|
+
detail=(
|
|
150
|
+
"the Prime CLI configuration on this machine could not be read "
|
|
151
|
+
"as a configuration, so no key can be resolved from it. Signing "
|
|
152
|
+
"in again writes a fresh one."
|
|
153
|
+
),
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
if state == "signed_out":
|
|
157
|
+
return CredentialStatus(
|
|
158
|
+
provider=model.provider,
|
|
159
|
+
credential_env=name,
|
|
160
|
+
available=False,
|
|
161
|
+
source="missing",
|
|
162
|
+
detail=(
|
|
163
|
+
"the Prime CLI configuration on this machine holds no key: this "
|
|
164
|
+
"machine is signed out, or the sign-in has been cleared. This "
|
|
165
|
+
"credential pays for the evaluated subject's model calls; it is "
|
|
166
|
+
"separate from whatever your own agent is signed in with."
|
|
167
|
+
),
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
return CredentialStatus(
|
|
171
|
+
provider=model.provider,
|
|
172
|
+
credential_env=name,
|
|
173
|
+
available=False,
|
|
174
|
+
source="missing",
|
|
175
|
+
detail=(
|
|
176
|
+
f"no active Prime CLI configuration supplies {name}. This credential "
|
|
177
|
+
"pays for the evaluated subject's model calls; it is separate from "
|
|
178
|
+
"whatever your own agent is signed in with."
|
|
179
|
+
),
|
|
180
|
+
)
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def require_credentials(model: ModelSpec) -> CredentialStatus:
|
|
184
|
+
"""Return the status, or refuse to go further without a credential."""
|
|
185
|
+
status = credential_status(model)
|
|
186
|
+
if status.available:
|
|
187
|
+
return status
|
|
188
|
+
raise AuthenticationError(
|
|
189
|
+
f"the evaluation model endpoint has no credential: {status.detail}",
|
|
190
|
+
code=MODEL_CREDENTIALS_MISSING,
|
|
191
|
+
details={
|
|
192
|
+
"provider": model.provider,
|
|
193
|
+
"model_id": model.model_id,
|
|
194
|
+
"credential_env": model.credential_env,
|
|
195
|
+
},
|
|
196
|
+
next_actions=[
|
|
197
|
+
NextAction(
|
|
198
|
+
id="sign_in_to_prime",
|
|
199
|
+
label="Sign in to Prime, then start the run again",
|
|
200
|
+
reason=(
|
|
201
|
+
"A PRIME_API_KEY-named credential resolves from the active "
|
|
202
|
+
"Prime CLI configuration, which a run can read for itself."
|
|
203
|
+
),
|
|
204
|
+
cli=["prime", "login"],
|
|
205
|
+
hermes_tool=None,
|
|
206
|
+
hermes_args=None,
|
|
207
|
+
requires_user_confirmation=True,
|
|
208
|
+
),
|
|
209
|
+
NextAction(
|
|
210
|
+
id="export_evaluation_credential",
|
|
211
|
+
label=(f"Check how {model.credential_env} reaches a run"),
|
|
212
|
+
reason=(
|
|
213
|
+
"Setting this credential in your own terminal is not "
|
|
214
|
+
"enough: a run works in a separate background process that "
|
|
215
|
+
"is not given your terminal's variables. It pays for the "
|
|
216
|
+
"evaluated subject's model calls and is never stored."
|
|
217
|
+
),
|
|
218
|
+
cli=["techtree", "doctor", "--for-evaluation"],
|
|
219
|
+
hermes_tool=None,
|
|
220
|
+
hermes_args=None,
|
|
221
|
+
requires_user_confirmation=False,
|
|
222
|
+
),
|
|
223
|
+
],
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _prime_config_state(home: str | None) -> _PrimeConfigState:
|
|
228
|
+
"""Report what the Prime CLI configuration under ``home`` supplies.
|
|
229
|
+
|
|
230
|
+
The file is opened, one key is tested for emptiness, and the value is
|
|
231
|
+
discarded. Nothing read here reaches a caller.
|
|
232
|
+
|
|
233
|
+
Three not-usable answers are distinguished because they need three
|
|
234
|
+
different sentences. No file is somebody who has not signed in; a file with
|
|
235
|
+
no key is somebody whose sign-in has gone away; a file that will not parse
|
|
236
|
+
is a broken store, and no amount of signing in explains itself if it is
|
|
237
|
+
described as either of the other two.
|
|
238
|
+
"""
|
|
239
|
+
if not home:
|
|
240
|
+
return "absent"
|
|
241
|
+
path = Path(home).joinpath(*PRIME_CONFIG_RELATIVE_PATH)
|
|
242
|
+
try:
|
|
243
|
+
raw = path.read_bytes()
|
|
244
|
+
except OSError:
|
|
245
|
+
return "absent"
|
|
246
|
+
try:
|
|
247
|
+
document = json.loads(raw)
|
|
248
|
+
except ValueError:
|
|
249
|
+
return "malformed"
|
|
250
|
+
if not isinstance(document, dict):
|
|
251
|
+
return "malformed"
|
|
252
|
+
value = document.get(_PRIME_CONFIG_KEY)
|
|
253
|
+
if value is None or (isinstance(value, str) and not value.strip()):
|
|
254
|
+
return "signed_out"
|
|
255
|
+
return "usable" if isinstance(value, str) else "malformed"
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def scrubbed_child_environment(
|
|
259
|
+
*,
|
|
260
|
+
model: ModelSpec,
|
|
261
|
+
engine: EngineInstallation,
|
|
262
|
+
extra: Mapping[str, str] | None = None,
|
|
263
|
+
) -> dict[str, str]:
|
|
264
|
+
"""Build a Verifiers child's environment from a narrow allow-list.
|
|
265
|
+
|
|
266
|
+
The host environment is not copied. A developer machine carries cloud
|
|
267
|
+
credentials, provider keys for other services, and shell configuration that
|
|
268
|
+
would change how the subject behaves, and none of it belongs inside an
|
|
269
|
+
experiment that claims only one thing differed.
|
|
270
|
+
|
|
271
|
+
The engine's own ``bin`` directory is prepended to ``PATH`` so the child
|
|
272
|
+
resolves the tools the pinned engine ships before anything the operator
|
|
273
|
+
happens to have installed, for the same reason the engine invokes its own
|
|
274
|
+
console scripts by absolute path.
|
|
275
|
+
"""
|
|
276
|
+
environment = {
|
|
277
|
+
name: os.environ[name] for name in _BASE_ENVIRONMENT if name in os.environ
|
|
278
|
+
}
|
|
279
|
+
environment["PATH"] = _engine_first_path(engine, environment.get("PATH"))
|
|
280
|
+
|
|
281
|
+
credential = os.environ.get(model.credential_env)
|
|
282
|
+
if credential:
|
|
283
|
+
environment[model.credential_env] = credential
|
|
284
|
+
|
|
285
|
+
for name in PRIME_ENVIRONMENT:
|
|
286
|
+
value = os.environ.get(name)
|
|
287
|
+
if value:
|
|
288
|
+
environment[name] = value
|
|
289
|
+
|
|
290
|
+
for name, value in (extra or {}).items():
|
|
291
|
+
environment[name] = value
|
|
292
|
+
|
|
293
|
+
return environment
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def _engine_first_path(engine: EngineInstallation, inherited: str | None) -> str:
|
|
297
|
+
"""Return a ``PATH`` that starts with the engine's own executables."""
|
|
298
|
+
engine_bin = str(Path(engine.python_executable).parent)
|
|
299
|
+
if not inherited:
|
|
300
|
+
return engine_bin
|
|
301
|
+
entries = [engine_bin, *(part for part in inherited.split(os.pathsep) if part)]
|
|
302
|
+
return os.pathsep.join(dict.fromkeys(entries))
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def redacted_environment(
|
|
306
|
+
environment: Mapping[str, str], *, secret_names: Iterable[str]
|
|
307
|
+
) -> dict[str, str]:
|
|
308
|
+
"""Describe a child environment safely enough to put in a diagnostic.
|
|
309
|
+
|
|
310
|
+
Named secrets are replaced by their length. That distinguishes an unset
|
|
311
|
+
variable from an empty one from a truncated paste, which is the whole of
|
|
312
|
+
what an operator needs, and discloses nothing usable. Everything else — the
|
|
313
|
+
``PATH`` the child searched, the ``HOME`` it read configuration from — is
|
|
314
|
+
shown, because hiding it would make the diagnostic useless without making
|
|
315
|
+
anything safer.
|
|
316
|
+
"""
|
|
317
|
+
secrets = set(secret_names)
|
|
318
|
+
return {
|
|
319
|
+
name: (f"<set, {len(value)} characters>" if name in secrets else value)
|
|
320
|
+
for name, value in environment.items()
|
|
321
|
+
}
|