techtree 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. techtree/__init__.py +35 -0
  2. techtree/__main__.py +14 -0
  3. techtree/canonical.py +239 -0
  4. techtree/catalog/__init__.py +25 -0
  5. techtree/catalog/repository.py +400 -0
  6. techtree/catalog/service.py +419 -0
  7. techtree/cli/__init__.py +1 -0
  8. techtree/cli/app.py +416 -0
  9. techtree/cli/commands/__init__.py +1 -0
  10. techtree/cli/commands/climb.py +1223 -0
  11. techtree/cli/commands/doctor.py +147 -0
  12. techtree/cli/commands/engine.py +207 -0
  13. techtree/cli/commands/proof.py +556 -0
  14. techtree/cli/commands/publish.py +447 -0
  15. techtree/cli/commands/release.py +303 -0
  16. techtree/cli/commands/run.py +1067 -0
  17. techtree/cli/commands/setup.py +181 -0
  18. techtree/cli/commands/skill.py +221 -0
  19. techtree/cli/commands/uplift.py +698 -0
  20. techtree/cli/commands/withdraw.py +212 -0
  21. techtree/cli/confirm.py +47 -0
  22. techtree/cli/context.py +96 -0
  23. techtree/cli/invoke.py +220 -0
  24. techtree/cli/output.py +280 -0
  25. techtree/constants.py +138 -0
  26. techtree/crypto.py +128 -0
  27. techtree/doctor/__init__.py +1 -0
  28. techtree/doctor/checks.py +675 -0
  29. techtree/doctor/execution_checks.py +435 -0
  30. techtree/doctor/service.py +326 -0
  31. techtree/drafts/__init__.py +32 -0
  32. techtree/drafts/source.py +146 -0
  33. techtree/drafts/store.py +992 -0
  34. techtree/engines/__init__.py +1 -0
  35. techtree/engines/bundle.py +251 -0
  36. techtree/engines/installer.py +679 -0
  37. techtree/engines/registry.py +235 -0
  38. techtree/engines/runner.py +170 -0
  39. techtree/errors.py +262 -0
  40. techtree/fs.py +234 -0
  41. techtree/harness.py +108 -0
  42. techtree/identity/__init__.py +41 -0
  43. techtree/identity/models.py +113 -0
  44. techtree/identity/service.py +199 -0
  45. techtree/identity/store.py +263 -0
  46. techtree/ids.py +85 -0
  47. techtree/manifests/__init__.py +39 -0
  48. techtree/manifests/builder.py +433 -0
  49. techtree/manifests/compare.py +376 -0
  50. techtree/models/__init__.py +282 -0
  51. techtree/models/base.py +201 -0
  52. techtree/models/campaign.py +484 -0
  53. techtree/models/catalog.py +227 -0
  54. techtree/models/cli.py +151 -0
  55. techtree/models/climb.py +254 -0
  56. techtree/models/data_policy.py +130 -0
  57. techtree/models/engine.py +156 -0
  58. techtree/models/episode_receipt.py +130 -0
  59. techtree/models/evaluation_backend.py +113 -0
  60. techtree/models/experiment.py +154 -0
  61. techtree/models/run.py +214 -0
  62. techtree/models/skill.py +156 -0
  63. techtree/models/uplift_report.py +158 -0
  64. techtree/models/validation.py +299 -0
  65. techtree/paths.py +116 -0
  66. techtree/presentation/__init__.py +31 -0
  67. techtree/presentation/build.py +1242 -0
  68. techtree/presentation/compact.py +246 -0
  69. techtree/presentation/evidence.py +169 -0
  70. techtree/presentation/models.py +358 -0
  71. techtree/presentation/rich.py +312 -0
  72. techtree/presentation/sanitize.py +156 -0
  73. techtree/publication/__init__.py +44 -0
  74. techtree/publication/address.py +180 -0
  75. techtree/publication/coordinates.py +26 -0
  76. techtree/publication/journal.py +212 -0
  77. techtree/publication/keccak.py +183 -0
  78. techtree/publication/models.py +209 -0
  79. techtree/publication/offer.py +35 -0
  80. techtree/publication/service.py +618 -0
  81. techtree/publication/transport.py +296 -0
  82. techtree/publication/verify.py +242 -0
  83. techtree/publication/withdraw.py +156 -0
  84. techtree/py.typed +0 -0
  85. techtree/receipts/__init__.py +52 -0
  86. techtree/receipts/bundle.py +578 -0
  87. techtree/receipts/compare.py +1065 -0
  88. techtree/receipts/episode.py +672 -0
  89. techtree/receipts/execution.py +630 -0
  90. techtree/receipts/observed.py +474 -0
  91. techtree/receipts/set.py +336 -0
  92. techtree/receipts/uplift.py +655 -0
  93. techtree/receipts/verify.py +1055 -0
  94. techtree/release/__init__.py +9 -0
  95. techtree/release/bootstrap.py +509 -0
  96. techtree/release/checks.py +376 -0
  97. techtree/release/document.py +125 -0
  98. techtree/release/generate.py +221 -0
  99. techtree/release/models.py +293 -0
  100. techtree/release/provenance.py +109 -0
  101. techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
  102. techtree/resources/catalog/catalog.json +32 -0
  103. techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
  104. techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
  105. techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
  106. techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
  107. techtree/resources/engines/default/engine.json +20 -0
  108. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
  109. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
  110. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
  111. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
  112. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
  113. techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
  114. techtree/resources/engines/default/pyproject.toml +23 -0
  115. techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
  116. techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
  117. techtree/resources/engines/default/tools/normalize_validation.py +222 -0
  118. techtree/resources/engines/default/uv.lock +1758 -0
  119. techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
  120. techtree/resources/release/build-provenance.json +4 -0
  121. techtree/resources/release/release-core.json +24 -0
  122. techtree/runs/__init__.py +31 -0
  123. techtree/runs/artifacts.py +750 -0
  124. techtree/runs/child_registry.py +228 -0
  125. techtree/runs/events.py +478 -0
  126. techtree/runs/executor.py +140 -0
  127. techtree/runs/fake.py +741 -0
  128. techtree/runs/launcher.py +253 -0
  129. techtree/runs/machine.py +489 -0
  130. techtree/runs/real.py +789 -0
  131. techtree/runs/service.py +616 -0
  132. techtree/runs/store.py +555 -0
  133. techtree/runs/validation.py +259 -0
  134. techtree/runs/variants.py +684 -0
  135. techtree/settings.py +143 -0
  136. techtree/skills/__init__.py +14 -0
  137. techtree/skills/archive.py +282 -0
  138. techtree/skills/policy.py +62 -0
  139. techtree/skills/scanner.py +394 -0
  140. techtree/skills/service.py +752 -0
  141. techtree/skills/starter.py +434 -0
  142. techtree/tasksets/__init__.py +1 -0
  143. techtree/tasksets/membership.py +269 -0
  144. techtree/tasksets/provider.py +207 -0
  145. techtree/tasksets/resolver.py +311 -0
  146. techtree/tasksets/service.py +484 -0
  147. techtree/tasksets/verifiers_cli.py +538 -0
  148. techtree/uplift/__init__.py +20 -0
  149. techtree/uplift/context.py +544 -0
  150. techtree/uplift/derive.py +203 -0
  151. techtree/uplift/public_tasks.py +151 -0
  152. techtree/uplift/service.py +719 -0
  153. techtree/uplift/source.py +160 -0
  154. techtree/verifiers/__init__.py +31 -0
  155. techtree/verifiers/budget.py +219 -0
  156. techtree/verifiers/child.py +633 -0
  157. techtree/verifiers/compiler.py +432 -0
  158. techtree/verifiers/config.py +365 -0
  159. techtree/verifiers/credentials.py +321 -0
  160. techtree/verifiers/image.py +126 -0
  161. techtree/verifiers/models.py +527 -0
  162. techtree/verifiers/outputs.py +368 -0
  163. techtree/verifiers/progress.py +192 -0
  164. techtree/verifiers/supervisor.py +341 -0
  165. techtree/verifiers/verify.py +782 -0
  166. techtree/version.py +39 -0
  167. techtree/worker/__init__.py +18 -0
  168. techtree/worker/execute.py +487 -0
  169. techtree/worker/main.py +57 -0
  170. techtree-0.1.0.dist-info/METADATA +344 -0
  171. techtree-0.1.0.dist-info/RECORD +174 -0
  172. techtree-0.1.0.dist-info/WHEEL +4 -0
  173. techtree-0.1.0.dist-info/entry_points.txt +3 -0
  174. techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,365 @@
1
+ """The only Verifiers configuration Techtree may emit. Spec section 6.7.
2
+
3
+ This module is an allow-list, and the allow-list is the control. Verifiers has
4
+ a large and growing set of knobs; a Campaign that could reach any of them could
5
+ introduce a second difference between baseline and candidate without anyone
6
+ being able to name it afterwards. Every model here forbids extra keys, so a
7
+ knob Techtree has not deliberately modelled is unrepresentable rather than
8
+ merely discouraged.
9
+
10
+ Several fields are typed as literals rather than validated at runtime, because
11
+ "the value is wrong" and "the value cannot be spelled" are different guarantees:
12
+
13
+ ``push: Literal[False]``
14
+ ``EvalConfig.push`` defaults to **true** upstream and uploads the complete
15
+ Episode — prompts and subject replies included — to the Prime platform
16
+ (``docs/verifiers-eval.md``, finding E1). A config that merely forgets to
17
+ set it exfiltrates the participant's trajectories.
18
+ ``rich: Literal[None]``
19
+ The dashboard is the whole output when it is on; a captured child needs log
20
+ lines. Upstream's ``rich`` is a table, not a flag, and **null is the only
21
+ spelling that turns the dashboard off**. Measured against the pinned build:
22
+ an omitted key resolves to ``{"show_logs": false}``, which is the dashboard
23
+ on *and* the log lines suppressed — the exact failure this lock-down
24
+ exists to prevent. So the danger here is a key that is missing, not a key
25
+ set to true, and :func:`emitted_document` writes this one null explicitly
26
+ while every other unset optional stays absent.
27
+ ``shuffle: Literal[False]`` and ``num_rollouts: Literal[1]``
28
+ Decisions document 0001. There is no seed anywhere in the protocol, so a
29
+ shuffled run could not be reproduced by anyone.
30
+ ``use_bundled_skill: Literal[False]``
31
+ A bundled skill catalogue is an uncontrolled second difference.
32
+
33
+ ``disabled_tools`` is absent by construction. The native Hermes harness accepts
34
+ it at config time and then refuses it in the middle of a run, after Docker has
35
+ been provisioned (``docs/verifiers-eval.md``, finding E3), so the only safe
36
+ place to reject it is here.
37
+
38
+ The Docker table carries ``allow`` and ``block``, which spec section 6.7 does
39
+ not model. Upstream's default egress policy is unrestricted, so without them a
40
+ Campaign declaring ``network_policy: "restricted"`` would compile to a
41
+ container with open network access.
42
+
43
+ The emitted document is **JSON**, and that is a safety property rather than a
44
+ taste. TOML has no null literal, so a TOML document cannot say "no dashboard"
45
+ at all — it can only leave ``rich`` out, which is the dangerous value. The
46
+ engine reads either format and picks its parser from the file extension, so
47
+ the configuration Techtree writes is named ``.json`` and every locked-down
48
+ setting stays expressible only as its safe value, in the file, where the run's
49
+ own inputs record it.
50
+ """
51
+
52
+ from __future__ import annotations
53
+
54
+ import json
55
+ import re
56
+ from typing import Any, Final, Literal, Self
57
+
58
+ from pydantic import BaseModel, ConfigDict, Field, model_validator
59
+
60
+ from techtree.errors import ValidationError
61
+ from techtree.models.campaign import CREDENTIAL_ENV_PATTERN
62
+
63
+ __all__ = [
64
+ "ALLOWED_CLIENT_HEADERS",
65
+ "EVAL_CONFIG_INVALID",
66
+ "IMAGE_DIGEST_PATTERN",
67
+ "OPEN_BLOCK",
68
+ "OPEN_NETWORK",
69
+ "RESTRICTED_BLOCK",
70
+ "RESTRICTED_NETWORK",
71
+ "DockerRuntimeToml",
72
+ "EnvToml",
73
+ "EvalClientToml",
74
+ "EvalToml",
75
+ "HermesHarnessToml",
76
+ "SamplingToml",
77
+ "SubjectAgentToml",
78
+ "TasksetToml",
79
+ "TimeoutToml",
80
+ "config_to_json_bytes",
81
+ "egress_for",
82
+ "emitted_document",
83
+ ]
84
+
85
+ #: Stable error code. Spec section 6.7.
86
+ EVAL_CONFIG_INVALID: Final = "eval_config_invalid"
87
+
88
+ #: Headers Techtree is allowed to declare on the evaluation client. Empty: the
89
+ #: only supported profile routes through the pinned client's own resolution,
90
+ #: and a header Techtree invents is a routing decision nobody reviewed. The
91
+ #: engine may still *add* headers of its own when it resolves the config
92
+ #: (``docs/verifiers-eval.md``); that is upstream's business, not Techtree's.
93
+ ALLOWED_CLIENT_HEADERS: Final[frozenset[str]] = frozenset()
94
+
95
+ #: Framework-only egress: the interception endpoint and nothing else, which is
96
+ #: what a restricted subject runtime means. Upstream normalizes an empty
97
+ #: allow-list to exactly this pair, so Techtree declares the normalized form
98
+ #: rather than the shorthand — otherwise the configuration Techtree compiled
99
+ #: and the configuration the engine resolved would disagree at ``block`` on
100
+ #: every restricted run.
101
+ RESTRICTED_NETWORK: Final[tuple[str, ...]] = ()
102
+ RESTRICTED_BLOCK: Final[tuple[str, ...]] = ("*",)
103
+ #: Unrestricted egress, upstream's own default spelling.
104
+ OPEN_NETWORK: Final[tuple[str, ...]] = ("*",)
105
+ OPEN_BLOCK: Final[tuple[str, ...]] = ()
106
+
107
+ #: An OCI reference that names content rather than a moving tag.
108
+ IMAGE_DIGEST_PATTERN: Final = r"@sha256:[0-9a-f]{64}$"
109
+
110
+ _CREDENTIAL_ENV_RE: Final = re.compile(CREDENTIAL_ENV_PATTERN)
111
+ _IMAGE_DIGEST_RE: Final = re.compile(IMAGE_DIGEST_PATTERN)
112
+
113
+ #: The settings whose null the emitted document must state out loud, because
114
+ #: leaving the key out would select something else. Only ``rich`` qualifies:
115
+ #: every other optional in this module defaults to the same ``None`` upstream.
116
+ _NULL_IS_THE_DECISION: Final[frozenset[str]] = frozenset({"rich"})
117
+
118
+
119
+ class TomlModel(BaseModel):
120
+ """A frozen, extra-forbidden fragment of the emitted configuration."""
121
+
122
+ model_config = ConfigDict(frozen=True, extra="forbid", validate_default=True)
123
+
124
+
125
+ class EvalClientToml(TomlModel):
126
+ """The OpenAI-compatible endpoint the evaluation runs against.
127
+
128
+ ``base_url`` is deliberately omitted for the supported profile. The pinned
129
+ client resolves a ``PRIME_API_KEY``-keyed endpoint from the environment and
130
+ the active Prime CLI configuration, and writing a URL here would freeze a
131
+ deployment detail into a run's inputs.
132
+ """
133
+
134
+ type: Literal["eval"] = "eval"
135
+ api_key_var: str
136
+ base_url: str | None = None
137
+ headers: dict[str, str] = Field(default_factory=dict)
138
+
139
+ @model_validator(mode="after")
140
+ def _check_the_client_names_a_variable_and_no_headers(self) -> Self:
141
+ """Reject a credential value, and any header Techtree may not declare."""
142
+ if _CREDENTIAL_ENV_RE.fullmatch(self.api_key_var) is None:
143
+ raise ValueError(
144
+ "api_key_var must be an uppercase environment-variable name, "
145
+ "never a credential value"
146
+ )
147
+ unknown = sorted(set(self.headers) - ALLOWED_CLIENT_HEADERS)
148
+ if unknown:
149
+ raise ValueError(f"Techtree does not declare client headers; got {unknown}")
150
+ return self
151
+
152
+
153
+ class SamplingToml(TomlModel):
154
+ """How the subject model is sampled."""
155
+
156
+ temperature: float = Field(ge=0.0, le=2.0)
157
+ max_tokens: int = Field(ge=1)
158
+
159
+
160
+ class HermesHarnessToml(TomlModel):
161
+ """The pinned Hermes Agent harness and the skills inserted into it."""
162
+
163
+ id: Literal["hermes-agent"] = "hermes-agent"
164
+ version: str = Field(min_length=1)
165
+ use_bundled_skill: Literal[False] = False
166
+ skills: list[str] = Field(default_factory=list)
167
+
168
+ @model_validator(mode="after")
169
+ def _check_skill_paths_are_absolute_and_distinct(self) -> Self:
170
+ """Reject a relative or repeated skill path."""
171
+ for path in self.skills:
172
+ if not path.startswith("/"):
173
+ raise ValueError(
174
+ f"skill paths are absolute run-owned paths; got {path!r}"
175
+ )
176
+ if len(set(self.skills)) != len(self.skills):
177
+ raise ValueError("a harness mounts each skill exactly once")
178
+ return self
179
+
180
+
181
+ class DockerRuntimeToml(TomlModel):
182
+ """Where the subject agent executes."""
183
+
184
+ type: Literal["docker"] = "docker"
185
+ image: str = Field(min_length=1)
186
+ allow: list[str] = Field(default_factory=list)
187
+ block: list[str] = Field(default_factory=list)
188
+ cpu: float | None = Field(default=None, gt=0.0)
189
+ memory: float | None = Field(default=None, gt=0.0)
190
+
191
+ @property
192
+ def image_is_digest_pinned(self) -> bool:
193
+ """Whether the image reference names content rather than a tag."""
194
+ return _IMAGE_DIGEST_RE.search(self.image) is not None
195
+
196
+ @property
197
+ def network_is_restricted(self) -> bool:
198
+ """Whether egress is anything narrower than unrestricted."""
199
+ return list(self.allow) != list(OPEN_NETWORK) or bool(self.block)
200
+
201
+ @model_validator(mode="after")
202
+ def _check_the_egress_lists_are_not_both_concrete(self) -> Self:
203
+ """Reject the one egress combination upstream refuses outright."""
204
+ if self.allow and list(self.allow) != list(OPEN_NETWORK) and self.block:
205
+ raise ValueError(
206
+ "a concrete allow list and a block list are mutually exclusive"
207
+ )
208
+ return self
209
+
210
+
211
+ class TimeoutToml(TomlModel):
212
+ """Per-rollout Verifiers lifecycle limits.
213
+
214
+ Every one of these defaults to ``None`` upstream, and ``None`` means "no
215
+ limit": the pinned build wraps each phase in ``asyncio.timeout(value)``, and
216
+ ``asyncio.timeout(None)`` is a no-op. A Campaign that declares
217
+ ``timeout_seconds`` and cannot reach this table has declared nothing, which
218
+ is the whole reason the table exists here (decisions document 0029, layer
219
+ A).
220
+ """
221
+
222
+ setup: float | None = Field(default=None, gt=0.0)
223
+ rollout: float | None = Field(default=None, gt=0.0)
224
+ finalize: float | None = Field(default=None, gt=0.0)
225
+ scoring: float | None = Field(default=None, gt=0.0)
226
+
227
+
228
+ class SubjectAgentToml(TomlModel):
229
+ """The one seat v0.1 evaluates, named for the role it plays."""
230
+
231
+ harness: HermesHarnessToml
232
+ runtime: DockerRuntimeToml
233
+ max_turns: int | None = Field(default=None, ge=1)
234
+ max_input_tokens: int | None = Field(default=None, ge=1)
235
+ max_output_tokens: int | None = Field(default=None, ge=1)
236
+ max_total_tokens: int | None = Field(default=None, ge=1)
237
+ timeout: TimeoutToml = Field(default_factory=TimeoutToml)
238
+
239
+
240
+ class TasksetToml(TomlModel):
241
+ """Which taskset the environment seeds from.
242
+
243
+ Only the identifier. Every other taskset field on the pinned reference
244
+ Taskset is a default Techtree does not vary, and a field Techtree does not
245
+ vary should not appear in a document a reader has to check.
246
+ """
247
+
248
+ id: str = Field(min_length=1)
249
+
250
+
251
+ class EnvToml(TomlModel):
252
+ """The environment block: the taskset, the subject seat, and the bound.
253
+
254
+ The seat is spelled ``subject`` because the reference package's ``Env``
255
+ declares a field of that name; Verifiers stamps the field name onto every
256
+ trace as ``agent.name`` (spec section 6.5). Against an environment without
257
+ that seat the whole configuration is rejected at parse time
258
+ (``docs/verifiers-eval.md``, finding E0).
259
+ """
260
+
261
+ taskset: TasksetToml
262
+ subject: SubjectAgentToml
263
+ max_concurrent_agents: int = Field(default=1, ge=1)
264
+
265
+
266
+ class EvalToml(TomlModel):
267
+ """One resolved Techtree experiment as a Verifiers evaluation."""
268
+
269
+ model: str = Field(min_length=1)
270
+ client: EvalClientToml
271
+ sampling: SamplingToml
272
+ env: EnvToml
273
+ num_tasks: int = Field(ge=1)
274
+ num_rollouts: Literal[1] = 1
275
+ shuffle: Literal[False] = False
276
+ max_concurrent: int = Field(ge=1)
277
+ rich: Literal[None] = None
278
+ push: Literal[False] = False
279
+ output_dir: str = Field(min_length=1)
280
+
281
+ @model_validator(mode="after")
282
+ def _check_the_run_is_bounded_and_run_owned(self) -> Self:
283
+ """Reject a relative output directory or an unbounded episode fan-out."""
284
+ if not self.output_dir.startswith("/"):
285
+ raise ValueError(
286
+ "output_dir is an absolute run-owned path; a relative path "
287
+ "lands wherever the child happened to be started"
288
+ )
289
+ if self.env.max_concurrent_agents > self.max_concurrent:
290
+ raise ValueError(
291
+ "max_concurrent_agents cannot exceed max_concurrent; the "
292
+ "product is the number of live subject runs"
293
+ )
294
+ return self
295
+
296
+
297
+ def egress_for(network_policy: str) -> tuple[list[str], list[str]]:
298
+ """Return the ``(allow, block)`` pair one Campaign network policy compiles to.
299
+
300
+ The Campaign speaks in intent — restricted or open — and upstream speaks in
301
+ two lists whose meaning depends on each other. This is the single place the
302
+ two vocabularies meet, and it emits the already-normalized form so nothing
303
+ downstream has to know that upstream would have rewritten the shorthand.
304
+ """
305
+ if network_policy == "restricted":
306
+ return list(RESTRICTED_NETWORK), list(RESTRICTED_BLOCK)
307
+ if network_policy == "open":
308
+ return list(OPEN_NETWORK), list(OPEN_BLOCK)
309
+ raise ValidationError(
310
+ f"{network_policy!r} is not a network policy Techtree can compile",
311
+ code=EVAL_CONFIG_INVALID,
312
+ details={"network_policy": network_policy},
313
+ )
314
+
315
+
316
+ def emitted_document(config: EvalToml) -> dict[str, Any]:
317
+ """Return exactly the mapping Techtree writes for one configuration.
318
+
319
+ ``exclude_none`` is kept for every field but one, because for every other
320
+ optional here a null and an absence mean the same thing to the engine and
321
+ an absence says it more honestly. ``client.base_url`` is left out so the
322
+ pinned client resolves the endpoint itself; an unset token ceiling or
323
+ timeout phase is a limit the Campaign did not declare, and upstream's own
324
+ default for each is the same ``None`` it would have been written as. In
325
+ every one of those cases the document should say only what Techtree
326
+ actually decided.
327
+
328
+ ``rich`` is the exception, and it is why the omission is not simply
329
+ switched off wholesale: its null is the decision, and dropping the key
330
+ selects the opposite (see the module docstring). So it is kept, which is
331
+ safe precisely because :class:`EvalToml` can spell no other value for it.
332
+ """
333
+ document: dict[str, Any] = config.model_dump(mode="json")
334
+ return {
335
+ key: _without_nulls(value)
336
+ for key, value in document.items()
337
+ if value is not None or key in _NULL_IS_THE_DECISION
338
+ }
339
+
340
+
341
+ def _without_nulls(value: Any) -> Any:
342
+ """Drop every unset optional from one nested value."""
343
+ if isinstance(value, dict):
344
+ return {
345
+ key: _without_nulls(inner)
346
+ for key, inner in value.items()
347
+ if inner is not None
348
+ }
349
+ return value
350
+
351
+
352
+ def config_to_json_bytes(config: EvalToml) -> bytes:
353
+ """Serialize one configuration to deterministic JSON bytes.
354
+
355
+ Key order is declaration order, so the same configuration always produces
356
+ the same bytes. The engine chooses its parser from the file's extension,
357
+ so these bytes belong in a file named ``.json``.
358
+ """
359
+ try:
360
+ return json.dumps(emitted_document(config), indent=2).encode("utf-8") + b"\n"
361
+ except (TypeError, ValueError) as error:
362
+ raise ValidationError(
363
+ f"the compiled evaluation config is not serializable: {error}",
364
+ code=EVAL_CONFIG_INVALID,
365
+ ) from error
@@ -0,0 +1,321 @@
1
+ """Whether the evaluation endpoint can authenticate. Spec section 6.9.
2
+
3
+ Two rules shape everything in this module.
4
+
5
+ *A secret is checked, never carried.* No function here returns a credential
6
+ value, writes one, or puts one in an error message, a detail dictionary, or a
7
+ log line. The only object that ever holds the value is the child process
8
+ environment :func:`scrubbed_child_environment` builds, and that dictionary is
9
+ handed straight to the child. :func:`redacted_environment` exists so that the
10
+ same dictionary can be described in a diagnostic without being disclosed.
11
+
12
+ *Evaluation auth is not host auth.* The credential this module diagnoses buys
13
+ model tokens for the evaluated subject. It is unrelated to whatever the
14
+ operator's own Hermes is authenticated with, and confusing the two produces the
15
+ worst possible failure: a run that looks configured, provisions Docker, and
16
+ then discovers thirty seconds later that nothing can answer. Spec sections 6.9
17
+ and 6.18 keep them separate, and so does the wording of every message here.
18
+
19
+ The pinned client resolves a ``PRIME_API_KEY``-named credential from the
20
+ environment first and from the active Prime CLI configuration second, returning
21
+ the literal string ``"EMPTY"`` when it finds neither
22
+ (``docs/verifiers-eval.md``). A missing credential therefore does not fail at
23
+ startup; it fails at the first model call, after the container is up. That is
24
+ the whole reason this check runs before a child is launched.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import json
30
+ import os
31
+ from collections.abc import Iterable, Mapping
32
+ from pathlib import Path
33
+ from typing import Final, Literal
34
+
35
+ from techtree.errors import AuthenticationError
36
+ from techtree.models.base import NonEmptyString, ProtocolModel
37
+ from techtree.models.campaign import ModelSpec
38
+ from techtree.models.cli import NextAction
39
+ from techtree.models.engine import EngineInstallation
40
+
41
+ __all__ = [
42
+ "MODEL_CREDENTIALS_MISSING",
43
+ "PRIME_CONFIG_RELATIVE_PATH",
44
+ "PRIME_CREDENTIAL_ENV",
45
+ "PRIME_ENVIRONMENT",
46
+ "CredentialStatus",
47
+ "credential_status",
48
+ "redacted_environment",
49
+ "require_credentials",
50
+ "scrubbed_child_environment",
51
+ ]
52
+
53
+ #: Stable error code. Spec section 6.9.
54
+ MODEL_CREDENTIALS_MISSING: Final = "model_credentials_missing"
55
+
56
+ #: The credential name the pinned client gives its own resolution to. Any other
57
+ #: name is read from the environment and nowhere else.
58
+ PRIME_CREDENTIAL_ENV: Final = "PRIME_API_KEY"
59
+
60
+ #: Where the Prime CLI keeps its configuration, relative to ``HOME``.
61
+ PRIME_CONFIG_RELATIVE_PATH: Final = (".prime", "config.json")
62
+
63
+ #: Prime variables the pinned client reads when it resolves an endpoint. The
64
+ #: key itself is not here: it is forwarded by name from the Campaign.
65
+ PRIME_ENVIRONMENT: Final[tuple[str, ...]] = (
66
+ "PRIME_INFERENCE_URL",
67
+ "PRIME_TEAM_ID",
68
+ )
69
+
70
+ #: The only host variables a child inherits. ``PATH`` so ordinary system tools
71
+ #: and the container runtime resolve, ``HOME`` because the Prime CLI
72
+ #: configuration and package caches hang off it, ``TMPDIR`` so scratch files
73
+ #: land where the host expects them.
74
+ _BASE_ENVIRONMENT: Final[tuple[str, ...]] = ("PATH", "HOME", "TMPDIR")
75
+
76
+ _PRIME_CONFIG_KEY: Final = "api_key"
77
+
78
+ #: What the Prime CLI configuration under one ``HOME`` supplies.
79
+ type _PrimeConfigState = Literal["usable", "signed_out", "malformed", "absent"]
80
+
81
+
82
+ class CredentialStatus(ProtocolModel):
83
+ """Whether one model endpoint can authenticate, and from where.
84
+
85
+ ``detail`` is operator-facing prose. It names the variable and the place it
86
+ was looked for, never a value or a fragment of one.
87
+
88
+ ``malformed_prime_config`` is its own answer rather than another way of
89
+ saying "missing": a configuration file that cannot be read is a broken
90
+ store, and telling somebody who has signed in that they have not is the
91
+ kind of advice that sends them round the loop again.
92
+ """
93
+
94
+ provider: NonEmptyString
95
+ credential_env: NonEmptyString
96
+ available: bool
97
+ source: Literal["environment", "prime_config", "missing", "malformed_prime_config"]
98
+ detail: NonEmptyString
99
+
100
+
101
+ def credential_status(
102
+ model: ModelSpec, *, environ: Mapping[str, str] | None = None
103
+ ) -> CredentialStatus:
104
+ """Report whether the declared credential can be resolved.
105
+
106
+ Presence only. The value is never read into a return, a log, or an error.
107
+
108
+ ``environ`` is the environment the question is asked *about*, which is not
109
+ always the one this process happens to have. A readiness check runs in an
110
+ operator's terminal but has to answer for the environment a detached run
111
+ would get, and a check that answered for its own terminal instead would be
112
+ able to say "ready" about a run that cannot authenticate.
113
+ """
114
+ source = os.environ if environ is None else environ
115
+ name = model.credential_env
116
+ if source.get(name):
117
+ return CredentialStatus(
118
+ provider=model.provider,
119
+ credential_env=name,
120
+ available=True,
121
+ source="environment",
122
+ detail=f"{name} is set in this environment.",
123
+ )
124
+
125
+ state = (
126
+ _prime_config_state(source.get("HOME"))
127
+ if name == PRIME_CREDENTIAL_ENV
128
+ else "absent"
129
+ )
130
+
131
+ if state == "usable":
132
+ return CredentialStatus(
133
+ provider=model.provider,
134
+ credential_env=name,
135
+ available=True,
136
+ source="prime_config",
137
+ detail=(
138
+ "the active Prime CLI configuration holds a key the pinned "
139
+ "evaluation client can use."
140
+ ),
141
+ )
142
+
143
+ if state == "malformed":
144
+ return CredentialStatus(
145
+ provider=model.provider,
146
+ credential_env=name,
147
+ available=False,
148
+ source="malformed_prime_config",
149
+ detail=(
150
+ "the Prime CLI configuration on this machine could not be read "
151
+ "as a configuration, so no key can be resolved from it. Signing "
152
+ "in again writes a fresh one."
153
+ ),
154
+ )
155
+
156
+ if state == "signed_out":
157
+ return CredentialStatus(
158
+ provider=model.provider,
159
+ credential_env=name,
160
+ available=False,
161
+ source="missing",
162
+ detail=(
163
+ "the Prime CLI configuration on this machine holds no key: this "
164
+ "machine is signed out, or the sign-in has been cleared. This "
165
+ "credential pays for the evaluated subject's model calls; it is "
166
+ "separate from whatever your own agent is signed in with."
167
+ ),
168
+ )
169
+
170
+ return CredentialStatus(
171
+ provider=model.provider,
172
+ credential_env=name,
173
+ available=False,
174
+ source="missing",
175
+ detail=(
176
+ f"no active Prime CLI configuration supplies {name}. This credential "
177
+ "pays for the evaluated subject's model calls; it is separate from "
178
+ "whatever your own agent is signed in with."
179
+ ),
180
+ )
181
+
182
+
183
+ def require_credentials(model: ModelSpec) -> CredentialStatus:
184
+ """Return the status, or refuse to go further without a credential."""
185
+ status = credential_status(model)
186
+ if status.available:
187
+ return status
188
+ raise AuthenticationError(
189
+ f"the evaluation model endpoint has no credential: {status.detail}",
190
+ code=MODEL_CREDENTIALS_MISSING,
191
+ details={
192
+ "provider": model.provider,
193
+ "model_id": model.model_id,
194
+ "credential_env": model.credential_env,
195
+ },
196
+ next_actions=[
197
+ NextAction(
198
+ id="sign_in_to_prime",
199
+ label="Sign in to Prime, then start the run again",
200
+ reason=(
201
+ "A PRIME_API_KEY-named credential resolves from the active "
202
+ "Prime CLI configuration, which a run can read for itself."
203
+ ),
204
+ cli=["prime", "login"],
205
+ hermes_tool=None,
206
+ hermes_args=None,
207
+ requires_user_confirmation=True,
208
+ ),
209
+ NextAction(
210
+ id="export_evaluation_credential",
211
+ label=(f"Check how {model.credential_env} reaches a run"),
212
+ reason=(
213
+ "Setting this credential in your own terminal is not "
214
+ "enough: a run works in a separate background process that "
215
+ "is not given your terminal's variables. It pays for the "
216
+ "evaluated subject's model calls and is never stored."
217
+ ),
218
+ cli=["techtree", "doctor", "--for-evaluation"],
219
+ hermes_tool=None,
220
+ hermes_args=None,
221
+ requires_user_confirmation=False,
222
+ ),
223
+ ],
224
+ )
225
+
226
+
227
+ def _prime_config_state(home: str | None) -> _PrimeConfigState:
228
+ """Report what the Prime CLI configuration under ``home`` supplies.
229
+
230
+ The file is opened, one key is tested for emptiness, and the value is
231
+ discarded. Nothing read here reaches a caller.
232
+
233
+ Three not-usable answers are distinguished because they need three
234
+ different sentences. No file is somebody who has not signed in; a file with
235
+ no key is somebody whose sign-in has gone away; a file that will not parse
236
+ is a broken store, and no amount of signing in explains itself if it is
237
+ described as either of the other two.
238
+ """
239
+ if not home:
240
+ return "absent"
241
+ path = Path(home).joinpath(*PRIME_CONFIG_RELATIVE_PATH)
242
+ try:
243
+ raw = path.read_bytes()
244
+ except OSError:
245
+ return "absent"
246
+ try:
247
+ document = json.loads(raw)
248
+ except ValueError:
249
+ return "malformed"
250
+ if not isinstance(document, dict):
251
+ return "malformed"
252
+ value = document.get(_PRIME_CONFIG_KEY)
253
+ if value is None or (isinstance(value, str) and not value.strip()):
254
+ return "signed_out"
255
+ return "usable" if isinstance(value, str) else "malformed"
256
+
257
+
258
+ def scrubbed_child_environment(
259
+ *,
260
+ model: ModelSpec,
261
+ engine: EngineInstallation,
262
+ extra: Mapping[str, str] | None = None,
263
+ ) -> dict[str, str]:
264
+ """Build a Verifiers child's environment from a narrow allow-list.
265
+
266
+ The host environment is not copied. A developer machine carries cloud
267
+ credentials, provider keys for other services, and shell configuration that
268
+ would change how the subject behaves, and none of it belongs inside an
269
+ experiment that claims only one thing differed.
270
+
271
+ The engine's own ``bin`` directory is prepended to ``PATH`` so the child
272
+ resolves the tools the pinned engine ships before anything the operator
273
+ happens to have installed, for the same reason the engine invokes its own
274
+ console scripts by absolute path.
275
+ """
276
+ environment = {
277
+ name: os.environ[name] for name in _BASE_ENVIRONMENT if name in os.environ
278
+ }
279
+ environment["PATH"] = _engine_first_path(engine, environment.get("PATH"))
280
+
281
+ credential = os.environ.get(model.credential_env)
282
+ if credential:
283
+ environment[model.credential_env] = credential
284
+
285
+ for name in PRIME_ENVIRONMENT:
286
+ value = os.environ.get(name)
287
+ if value:
288
+ environment[name] = value
289
+
290
+ for name, value in (extra or {}).items():
291
+ environment[name] = value
292
+
293
+ return environment
294
+
295
+
296
+ def _engine_first_path(engine: EngineInstallation, inherited: str | None) -> str:
297
+ """Return a ``PATH`` that starts with the engine's own executables."""
298
+ engine_bin = str(Path(engine.python_executable).parent)
299
+ if not inherited:
300
+ return engine_bin
301
+ entries = [engine_bin, *(part for part in inherited.split(os.pathsep) if part)]
302
+ return os.pathsep.join(dict.fromkeys(entries))
303
+
304
+
305
+ def redacted_environment(
306
+ environment: Mapping[str, str], *, secret_names: Iterable[str]
307
+ ) -> dict[str, str]:
308
+ """Describe a child environment safely enough to put in a diagnostic.
309
+
310
+ Named secrets are replaced by their length. That distinguishes an unset
311
+ variable from an empty one from a truncated paste, which is the whole of
312
+ what an operator needs, and discloses nothing usable. Everything else — the
313
+ ``PATH`` the child searched, the ``HOME`` it read configuration from — is
314
+ shown, because hiding it would make the diagnostic useless without making
315
+ anything safer.
316
+ """
317
+ secrets = set(secret_names)
318
+ return {
319
+ name: (f"<set, {len(value)} characters>" if name in secrets else value)
320
+ for name, value in environment.items()
321
+ }