techtree 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. techtree/__init__.py +35 -0
  2. techtree/__main__.py +14 -0
  3. techtree/canonical.py +239 -0
  4. techtree/catalog/__init__.py +25 -0
  5. techtree/catalog/repository.py +400 -0
  6. techtree/catalog/service.py +419 -0
  7. techtree/cli/__init__.py +1 -0
  8. techtree/cli/app.py +416 -0
  9. techtree/cli/commands/__init__.py +1 -0
  10. techtree/cli/commands/climb.py +1223 -0
  11. techtree/cli/commands/doctor.py +147 -0
  12. techtree/cli/commands/engine.py +207 -0
  13. techtree/cli/commands/proof.py +556 -0
  14. techtree/cli/commands/publish.py +447 -0
  15. techtree/cli/commands/release.py +303 -0
  16. techtree/cli/commands/run.py +1067 -0
  17. techtree/cli/commands/setup.py +181 -0
  18. techtree/cli/commands/skill.py +221 -0
  19. techtree/cli/commands/uplift.py +698 -0
  20. techtree/cli/commands/withdraw.py +212 -0
  21. techtree/cli/confirm.py +47 -0
  22. techtree/cli/context.py +96 -0
  23. techtree/cli/invoke.py +220 -0
  24. techtree/cli/output.py +280 -0
  25. techtree/constants.py +138 -0
  26. techtree/crypto.py +128 -0
  27. techtree/doctor/__init__.py +1 -0
  28. techtree/doctor/checks.py +675 -0
  29. techtree/doctor/execution_checks.py +435 -0
  30. techtree/doctor/service.py +326 -0
  31. techtree/drafts/__init__.py +32 -0
  32. techtree/drafts/source.py +146 -0
  33. techtree/drafts/store.py +992 -0
  34. techtree/engines/__init__.py +1 -0
  35. techtree/engines/bundle.py +251 -0
  36. techtree/engines/installer.py +679 -0
  37. techtree/engines/registry.py +235 -0
  38. techtree/engines/runner.py +170 -0
  39. techtree/errors.py +262 -0
  40. techtree/fs.py +234 -0
  41. techtree/harness.py +108 -0
  42. techtree/identity/__init__.py +41 -0
  43. techtree/identity/models.py +113 -0
  44. techtree/identity/service.py +199 -0
  45. techtree/identity/store.py +263 -0
  46. techtree/ids.py +85 -0
  47. techtree/manifests/__init__.py +39 -0
  48. techtree/manifests/builder.py +433 -0
  49. techtree/manifests/compare.py +376 -0
  50. techtree/models/__init__.py +282 -0
  51. techtree/models/base.py +201 -0
  52. techtree/models/campaign.py +484 -0
  53. techtree/models/catalog.py +227 -0
  54. techtree/models/cli.py +151 -0
  55. techtree/models/climb.py +254 -0
  56. techtree/models/data_policy.py +130 -0
  57. techtree/models/engine.py +156 -0
  58. techtree/models/episode_receipt.py +130 -0
  59. techtree/models/evaluation_backend.py +113 -0
  60. techtree/models/experiment.py +154 -0
  61. techtree/models/run.py +214 -0
  62. techtree/models/skill.py +156 -0
  63. techtree/models/uplift_report.py +158 -0
  64. techtree/models/validation.py +299 -0
  65. techtree/paths.py +116 -0
  66. techtree/presentation/__init__.py +31 -0
  67. techtree/presentation/build.py +1242 -0
  68. techtree/presentation/compact.py +246 -0
  69. techtree/presentation/evidence.py +169 -0
  70. techtree/presentation/models.py +358 -0
  71. techtree/presentation/rich.py +312 -0
  72. techtree/presentation/sanitize.py +156 -0
  73. techtree/publication/__init__.py +44 -0
  74. techtree/publication/address.py +180 -0
  75. techtree/publication/coordinates.py +26 -0
  76. techtree/publication/journal.py +212 -0
  77. techtree/publication/keccak.py +183 -0
  78. techtree/publication/models.py +209 -0
  79. techtree/publication/offer.py +35 -0
  80. techtree/publication/service.py +618 -0
  81. techtree/publication/transport.py +296 -0
  82. techtree/publication/verify.py +242 -0
  83. techtree/publication/withdraw.py +156 -0
  84. techtree/py.typed +0 -0
  85. techtree/receipts/__init__.py +52 -0
  86. techtree/receipts/bundle.py +578 -0
  87. techtree/receipts/compare.py +1065 -0
  88. techtree/receipts/episode.py +672 -0
  89. techtree/receipts/execution.py +630 -0
  90. techtree/receipts/observed.py +474 -0
  91. techtree/receipts/set.py +336 -0
  92. techtree/receipts/uplift.py +655 -0
  93. techtree/receipts/verify.py +1055 -0
  94. techtree/release/__init__.py +9 -0
  95. techtree/release/bootstrap.py +509 -0
  96. techtree/release/checks.py +376 -0
  97. techtree/release/document.py +125 -0
  98. techtree/release/generate.py +221 -0
  99. techtree/release/models.py +293 -0
  100. techtree/release/provenance.py +109 -0
  101. techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
  102. techtree/resources/catalog/catalog.json +32 -0
  103. techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
  104. techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
  105. techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
  106. techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
  107. techtree/resources/engines/default/engine.json +20 -0
  108. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
  109. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
  110. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
  111. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
  112. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
  113. techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
  114. techtree/resources/engines/default/pyproject.toml +23 -0
  115. techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
  116. techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
  117. techtree/resources/engines/default/tools/normalize_validation.py +222 -0
  118. techtree/resources/engines/default/uv.lock +1758 -0
  119. techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
  120. techtree/resources/release/build-provenance.json +4 -0
  121. techtree/resources/release/release-core.json +24 -0
  122. techtree/runs/__init__.py +31 -0
  123. techtree/runs/artifacts.py +750 -0
  124. techtree/runs/child_registry.py +228 -0
  125. techtree/runs/events.py +478 -0
  126. techtree/runs/executor.py +140 -0
  127. techtree/runs/fake.py +741 -0
  128. techtree/runs/launcher.py +253 -0
  129. techtree/runs/machine.py +489 -0
  130. techtree/runs/real.py +789 -0
  131. techtree/runs/service.py +616 -0
  132. techtree/runs/store.py +555 -0
  133. techtree/runs/validation.py +259 -0
  134. techtree/runs/variants.py +684 -0
  135. techtree/settings.py +143 -0
  136. techtree/skills/__init__.py +14 -0
  137. techtree/skills/archive.py +282 -0
  138. techtree/skills/policy.py +62 -0
  139. techtree/skills/scanner.py +394 -0
  140. techtree/skills/service.py +752 -0
  141. techtree/skills/starter.py +434 -0
  142. techtree/tasksets/__init__.py +1 -0
  143. techtree/tasksets/membership.py +269 -0
  144. techtree/tasksets/provider.py +207 -0
  145. techtree/tasksets/resolver.py +311 -0
  146. techtree/tasksets/service.py +484 -0
  147. techtree/tasksets/verifiers_cli.py +538 -0
  148. techtree/uplift/__init__.py +20 -0
  149. techtree/uplift/context.py +544 -0
  150. techtree/uplift/derive.py +203 -0
  151. techtree/uplift/public_tasks.py +151 -0
  152. techtree/uplift/service.py +719 -0
  153. techtree/uplift/source.py +160 -0
  154. techtree/verifiers/__init__.py +31 -0
  155. techtree/verifiers/budget.py +219 -0
  156. techtree/verifiers/child.py +633 -0
  157. techtree/verifiers/compiler.py +432 -0
  158. techtree/verifiers/config.py +365 -0
  159. techtree/verifiers/credentials.py +321 -0
  160. techtree/verifiers/image.py +126 -0
  161. techtree/verifiers/models.py +527 -0
  162. techtree/verifiers/outputs.py +368 -0
  163. techtree/verifiers/progress.py +192 -0
  164. techtree/verifiers/supervisor.py +341 -0
  165. techtree/verifiers/verify.py +782 -0
  166. techtree/version.py +39 -0
  167. techtree/worker/__init__.py +18 -0
  168. techtree/worker/execute.py +487 -0
  169. techtree/worker/main.py +57 -0
  170. techtree-0.1.0.dist-info/METADATA +344 -0
  171. techtree-0.1.0.dist-info/RECORD +174 -0
  172. techtree-0.1.0.dist-info/WHEEL +4 -0
  173. techtree-0.1.0.dist-info/entry_points.txt +3 -0
  174. techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,126 @@
1
+ """Asking the local daemon what the pinned subject image is. Decisions 0007 R5.
2
+
3
+ The evaluation cannot answer this. The pinned Verifiers build hands Docker an
4
+ image reference and records the reference, so its own output says what was
5
+ *asked for* and nothing about what was *there*. A comparison built on that alone
6
+ can only report the Campaign's own pin back to the reader, which is why the
7
+ image digest used to be a warning rather than a check.
8
+
9
+ So Techtree asks, once per variant, immediately before that variant's child is
10
+ launched. Two facts come back, both from the daemon:
11
+
12
+ *the content it holds* — ``docker image inspect`` resolves the pinned reference
13
+ or fails. A daemon that resolves ``repository@sha256:...`` is holding exactly
14
+ that content, and the repository digests it lists for the image are required to
15
+ include the pin, so a resolution that came from somewhere else is refused;
16
+
17
+ *the platform it serves* — the operating system and architecture the index was
18
+ resolved to on this host. The platform-specific manifest digest is *not* asked
19
+ of the daemon, because the daemon does not know it: an image pulled by index
20
+ digest keeps the index digest and the unpacked platform image, not the platform
21
+ manifest's own digest. That digest is a property of the pinned index, recorded
22
+ per platform in the Campaign, and read out by the platform observed here.
23
+
24
+ Nothing here pulls. Provisioning an image is an explicit setup step the operator
25
+ asks for (spec section 6.18), and an executor that downloaded a few hundred
26
+ megabytes to make its own check pass would be spending somebody's bandwidth to
27
+ avoid telling them the truth.
28
+ """
29
+
30
+ from __future__ import annotations
31
+
32
+ import json
33
+ import shutil
34
+ import subprocess
35
+ from collections.abc import Iterable
36
+ from typing import Final
37
+
38
+ from techtree.errors import ValidationError
39
+ from techtree.models.base import JsonValue
40
+ from techtree.models.campaign import RuntimeSpec
41
+ from techtree.verifiers.models import SubjectImageResolution, VariantName
42
+
43
+ __all__ = [
44
+ "IMAGE_INSPECT_TIMEOUT_SECONDS",
45
+ "SUBJECT_IMAGE_UNRESOLVED",
46
+ "resolve_subject_image",
47
+ ]
48
+
49
+ #: Stable error code. Spec section 15.
50
+ SUBJECT_IMAGE_UNRESOLVED: Final = "subject_image_unresolved"
51
+
52
+ #: Inspecting a local image is a metadata read, but the daemon may still be
53
+ #: waking up.
54
+ IMAGE_INSPECT_TIMEOUT_SECONDS: Final = 30.0
55
+
56
+ _INSPECT_FORMAT: Final = "{{json .RepoDigests}}\t{{.Os}}/{{.Architecture}}"
57
+
58
+
59
+ def resolve_subject_image(
60
+ runtime: RuntimeSpec, variant: VariantName
61
+ ) -> SubjectImageResolution:
62
+ """Return what this machine's daemon holds for the Campaign's subject image."""
63
+ if shutil.which("docker") is None:
64
+ raise ValidationError(
65
+ "docker is not on PATH, so what the subject container would run "
66
+ "cannot be established",
67
+ code=SUBJECT_IMAGE_UNRESOLVED,
68
+ details={"variant": variant.value, "image": runtime.image},
69
+ )
70
+
71
+ completed = subprocess.run(
72
+ ["docker", "image", "inspect", runtime.image, "--format", _INSPECT_FORMAT],
73
+ capture_output=True,
74
+ text=True,
75
+ check=False,
76
+ timeout=IMAGE_INSPECT_TIMEOUT_SECONDS,
77
+ stdin=subprocess.DEVNULL,
78
+ )
79
+ if completed.returncode != 0 or not completed.stdout.strip():
80
+ raise ValidationError(
81
+ f"the Docker daemon does not hold {runtime.image}; pull it as an "
82
+ "explicit setup step before running an evaluation",
83
+ code=SUBJECT_IMAGE_UNRESOLVED,
84
+ details={
85
+ "variant": variant.value,
86
+ "image": runtime.image,
87
+ "exit_code": completed.returncode,
88
+ },
89
+ )
90
+
91
+ digests, _, platform = completed.stdout.strip().partition("\t")
92
+ repository_digests = json.loads(digests)
93
+ if runtime.image not in repository_digests:
94
+ raise ValidationError(
95
+ "the image the daemon resolved does not list the content the "
96
+ "Campaign pinned among its own repository digests",
97
+ code=SUBJECT_IMAGE_UNRESOLVED,
98
+ details={
99
+ "variant": variant.value,
100
+ "image": runtime.image,
101
+ "repository_digests": _text_detail(repository_digests),
102
+ },
103
+ )
104
+ if platform not in runtime.image_platform_digests:
105
+ raise ValidationError(
106
+ f"the daemon serves {runtime.image} as {platform}, which the "
107
+ "Campaign pins no manifest digest for",
108
+ code=SUBJECT_IMAGE_UNRESOLVED,
109
+ details={
110
+ "variant": variant.value,
111
+ "platform": platform,
112
+ "pinned_platforms": _text_detail(runtime.image_platform_digests),
113
+ },
114
+ )
115
+
116
+ return SubjectImageResolution(
117
+ variant=variant,
118
+ image=runtime.image,
119
+ index_digest=runtime.image_index_digest,
120
+ platform=platform,
121
+ )
122
+
123
+
124
+ def _text_detail(values: Iterable[object]) -> list[JsonValue]:
125
+ """Return an ordered, printable list in the shape error details carry."""
126
+ return [text for text in sorted(str(value) for value in values)]
@@ -0,0 +1,527 @@
1
+ """Local integration types for native Verifiers execution. Spec section 6.6.
2
+
3
+ Nothing in this module is a Techtree protocol root. These objects describe one
4
+ machine's execution of one Campaign: which child ran, where its bytes landed,
5
+ and what the pinned engine's normalizer made of them. They are hashed and
6
+ written into a run directory, never into the Campaign graph and never onto a
7
+ website.
8
+
9
+ ``RunPaths`` is the one addition the specification names but the repository did
10
+ not yet have (spec section 6.19). It exists here rather than in
11
+ ``techtree.paths`` because every path it owns belongs to Verifiers execution;
12
+ ``TechtreePaths`` stays the answer to "where does Techtree keep its state", and
13
+ this stays the answer to "where does one variant's evaluation live".
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from dataclasses import dataclass
19
+ from datetime import datetime
20
+ from enum import StrEnum
21
+ from pathlib import Path
22
+ from typing import Final, Literal, Self
23
+
24
+ from pydantic import Field, model_validator
25
+
26
+ from techtree.models.base import (
27
+ ArtifactRef,
28
+ Digest,
29
+ JsonValue,
30
+ NonEmptyString,
31
+ ProtocolModel,
32
+ )
33
+ from techtree.models.campaign import VariantSchedule
34
+ from techtree.paths import TechtreePaths
35
+
36
+ __all__ = [
37
+ "COMMAND_LOG_FILENAME",
38
+ "EVAL_RUN_NAME",
39
+ "INPUT_CONFIG_FILENAME",
40
+ "NORMALIZED_EPISODES_FILENAME",
41
+ "STDERR_LOG_FILENAME",
42
+ "STDOUT_LOG_FILENAME",
43
+ "SUPERVISION_RECORD_FILENAME",
44
+ "VERIFIERS_DIRECTORY",
45
+ "ChildProcessOutcome",
46
+ "ExecutionCheck",
47
+ "NormalizedEpisode",
48
+ "NormalizedExecutionError",
49
+ "NormalizedReward",
50
+ "NormalizedRuntime",
51
+ "NormalizedTool",
52
+ "NormalizedTrace",
53
+ "NormalizedUsage",
54
+ "RealExecutionResult",
55
+ "RunPaths",
56
+ "SubjectImageResolution",
57
+ "VariantExecutionPlan",
58
+ "VariantExecutionResult",
59
+ "VariantName",
60
+ ]
61
+
62
+ #: The run-owned subtree every variant's evaluation lives under.
63
+ VERIFIERS_DIRECTORY: Final = "verifiers"
64
+ #: The configuration Techtree compiles, as opposed to the one the engine
65
+ #: resolves and writes back out under its own name. The extension is load
66
+ #: bearing rather than cosmetic: the engine chooses its parser from it, and
67
+ #: JSON is the only one of the two formats that can spell the explicit null
68
+ #: that turns the live dashboard off (``src/techtree/verifiers/config.py``).
69
+ INPUT_CONFIG_FILENAME: Final = "input.json"
70
+
71
+ #: What the engine writes when it normalizes one variant's raw episodes. It sits
72
+ #: beside the raw evidence inside ``run/`` (spec section 6.19) rather than above
73
+ #: it, so that one directory holds everything one execution produced.
74
+ NORMALIZED_EPISODES_FILENAME: Final = "normalized-episodes.jsonl"
75
+
76
+ #: Where a live child's captured streams land. Spec section 6.10 redirects both
77
+ #: to run-owned files; section 6.19 names them.
78
+ STDOUT_LOG_FILENAME: Final = "stdout.log"
79
+ STDERR_LOG_FILENAME: Final = "stderr.log"
80
+
81
+ #: The dry run is short and captured whole, so it is recorded in one file rather
82
+ #: than as a pair of streams. Spec section 6.19.
83
+ COMMAND_LOG_FILENAME: Final = "command.log"
84
+
85
+ #: What one variant's supervisor leaves behind: why the evaluation ended, and
86
+ #: how long it took to stop (decisions document 0029, layer B). It sits beside
87
+ #: the evaluation rather than inside ``run/`` because the engine owns that
88
+ #: directory and this is Techtree's own record of the engine's lifetime.
89
+ SUPERVISION_RECORD_FILENAME: Final = "supervision.json"
90
+
91
+ #: What Techtree calls the one evaluation it groups under a variant's output
92
+ #: directory. Since v0.3.1 ``--output-dir`` names the directory runs are
93
+ #: *grouped* under and the run itself lands in ``<output-dir>/<run.dir>``, with
94
+ #: a random suffix when nothing names it (``docs/verifiers-pin-0.3.1.md``,
95
+ #: deviation D2). Naming it is what makes one variant's evidence findable, and
96
+ #: findable twice.
97
+ EVAL_RUN_NAME: Final = "run"
98
+
99
+ _DRY_RUN_DIRECTORY: Final = "dry-run"
100
+ _INPUTS_DIRECTORY: Final = "inputs"
101
+ _MANIFESTS_DIRECTORY: Final = "manifests"
102
+ _SKILL_FILES_PATH: Final = ("skill", "files")
103
+
104
+
105
+ class VariantName(StrEnum):
106
+ """Which side of the comparison a child process is running."""
107
+
108
+ BASELINE = "baseline"
109
+ CANDIDATE = "candidate"
110
+
111
+
112
+ # ---------------------------------------------------------------------------
113
+ # Where one run's evaluation lives
114
+ # ---------------------------------------------------------------------------
115
+
116
+
117
+ @dataclass(frozen=True)
118
+ class RunPaths:
119
+ """Every path one run's Verifiers execution is allowed to touch.
120
+
121
+ Spec section 6.19. The layout is per variant so that a parallel schedule
122
+ cannot have two children writing the same file, and so that a cancelled
123
+ variant's partial evidence stays legible next to its sibling's complete
124
+ evidence.
125
+ """
126
+
127
+ root: Path
128
+
129
+ @classmethod
130
+ def for_run(cls, paths: TechtreePaths, run_id: str) -> Self:
131
+ """Locate one run's directory inside a Techtree home."""
132
+ return cls(root=paths.run_dir(run_id))
133
+
134
+ @property
135
+ def inputs_dir(self) -> Path:
136
+ """The run's own copies of everything it executes."""
137
+ return self.root / _INPUTS_DIRECTORY
138
+
139
+ @property
140
+ def skill_files_dir(self) -> Path:
141
+ """The run-owned skill tree a compiled config may point at."""
142
+ return self.inputs_dir.joinpath(*_SKILL_FILES_PATH)
143
+
144
+ def manifest_path(self, variant: VariantName) -> Path:
145
+ """The run's own copy of one variant's experiment manifest."""
146
+ return self.inputs_dir / _MANIFESTS_DIRECTORY / f"{variant.value}.json"
147
+
148
+ @property
149
+ def verifiers_dir(self) -> Path:
150
+ """The root of every variant's evaluation."""
151
+ return self.root / VERIFIERS_DIRECTORY
152
+
153
+ def variant_dir(self, variant: VariantName) -> Path:
154
+ """One variant's evaluation directory."""
155
+ return self.verifiers_dir / variant.value
156
+
157
+ def variant_input_config(self, variant: VariantName) -> Path:
158
+ """The configuration Techtree compiles for one variant."""
159
+ return self.variant_dir(variant) / INPUT_CONFIG_FILENAME
160
+
161
+ def variant_dry_run_dir(self, variant: VariantName) -> Path:
162
+ """Where the engine writes the resolved config during validation.
163
+
164
+ Separate from the run directory on purpose: a dry run writes only the
165
+ resolved configuration (``docs/verifiers-eval.md``, finding E2), and
166
+ letting it land beside real evidence would leave a directory that looks
167
+ like a truncated run.
168
+ """
169
+ return self.variant_dir(variant) / _DRY_RUN_DIRECTORY
170
+
171
+ def variant_dry_run_command_log(self, variant: VariantName) -> Path:
172
+ """What the dry-run invocation was, and what it said back."""
173
+ return self.variant_dry_run_dir(variant) / COMMAND_LOG_FILENAME
174
+
175
+ def variant_supervision_record(self, variant: VariantName) -> Path:
176
+ """Where one variant's supervisor records how its evaluation ended."""
177
+ return self.variant_dir(variant) / SUPERVISION_RECORD_FILENAME
178
+
179
+ def variant_output_group_dir(self, variant: VariantName) -> Path:
180
+ """The directory the engine groups one variant's evaluations under.
181
+
182
+ This is what the compiled configuration's ``output_dir`` names, and it
183
+ is a level above the run itself: ``--output-dir`` groups runs rather
184
+ than receiving one (deviation D2).
185
+ """
186
+ return self.variant_dir(variant)
187
+
188
+ def variant_output_dir(self, variant: VariantName) -> Path:
189
+ """Where the engine writes one variant's real evaluation output.
190
+
191
+ One level below the group directory, under the name the invocation
192
+ pins with ``--run.name``. Left unpinned the engine would append a
193
+ random suffix here and the evidence would land somewhere Techtree
194
+ never looks.
195
+ """
196
+ return self.variant_output_group_dir(variant) / EVAL_RUN_NAME
197
+
198
+ def variant_stdout_log(self, variant: VariantName) -> Path:
199
+ """Where one variant's child sends everything it prints.
200
+
201
+ Never a console. With ``rich`` disabled the pinned CLI dumps every
202
+ trace as indented JSON when the run ends, and those are the subject's
203
+ transcripts (``docs/verifiers-eval.md``).
204
+ """
205
+ return self.variant_output_dir(variant) / STDOUT_LOG_FILENAME
206
+
207
+ def variant_stderr_log(self, variant: VariantName) -> Path:
208
+ """Where one variant's child sends its diagnostics."""
209
+ return self.variant_output_dir(variant) / STDERR_LOG_FILENAME
210
+
211
+ def variant_normalized_episodes(self, variant: VariantName) -> Path:
212
+ """One variant's normalized projection, beside the evidence it projects."""
213
+ return self.variant_output_dir(variant) / NORMALIZED_EPISODES_FILENAME
214
+
215
+ def relative(self, path: Path) -> str:
216
+ """Return ``path`` as a POSIX path relative to the run directory."""
217
+ return path.relative_to(self.root).as_posix()
218
+
219
+ def owns(self, path: Path) -> bool:
220
+ """Whether ``path`` lies inside the run's own input tree."""
221
+ return path.is_absolute() and path.is_relative_to(self.inputs_dir)
222
+
223
+
224
+ # ---------------------------------------------------------------------------
225
+ # Planning and process outcome
226
+ # ---------------------------------------------------------------------------
227
+
228
+
229
+ class VariantExecutionPlan(ProtocolModel):
230
+ """Everything one variant's child process needs, resolved."""
231
+
232
+ variant: VariantName
233
+ experiment_manifest_digest: Digest
234
+ experiment_manifest_path: NonEmptyString
235
+ verifiers_input_config_path: NonEmptyString
236
+ verifiers_output_dir: NonEmptyString
237
+ skill_paths: list[NonEmptyString]
238
+ task_count: int = Field(ge=1)
239
+ max_concurrent: int = Field(ge=1)
240
+
241
+
242
+ class SubjectImageResolution(ProtocolModel):
243
+ """What the local daemon answered about the pinned subject image.
244
+
245
+ The pinned Verifiers build asks Docker for a reference and records only the
246
+ reference, so the evaluation's own output cannot say what the daemon held or
247
+ which platform it served. Techtree asks, once per variant, immediately
248
+ before that variant's child is launched, and records the answer here.
249
+
250
+ Two facts, both from the daemon: the content digest it holds for the
251
+ reference — for a multi-platform repository, the OCI image index — and the
252
+ platform it resolved that index to on this host. The platform-specific
253
+ manifest digest is not asked of the daemon because the daemon does not know
254
+ it; it is a property of the pinned index, recorded in the Campaign per
255
+ supported platform, and the comparison reads it out by the platform observed
256
+ here.
257
+ """
258
+
259
+ variant: VariantName
260
+ image: NonEmptyString
261
+ index_digest: Digest
262
+ platform: NonEmptyString
263
+
264
+
265
+ class ChildProcessOutcome(ProtocolModel):
266
+ """What one Verifiers child process did.
267
+
268
+ ``argv_digest`` rather than the argv itself. The compiled invocation never
269
+ carries a secret, but a digest is what makes that claim checkable without
270
+ re-reading a command line into a log.
271
+ """
272
+
273
+ variant: VariantName
274
+ argv_digest: Digest
275
+ exit_code: int
276
+ started_at: datetime
277
+ finished_at: datetime
278
+ stdout_artifact: ArtifactRef
279
+ stderr_artifact: ArtifactRef
280
+ cancelled: bool
281
+
282
+ @model_validator(mode="after")
283
+ def _check_the_clock_moves_forward(self) -> Self:
284
+ """Reject a process that finished before it started."""
285
+ if self.finished_at < self.started_at:
286
+ raise ValueError("a child process cannot finish before it starts")
287
+ return self
288
+
289
+
290
+ # ---------------------------------------------------------------------------
291
+ # The normalized projection of upstream evidence
292
+ # ---------------------------------------------------------------------------
293
+
294
+
295
+ class NormalizedExecutionError(ProtocolModel):
296
+ """One failure the engine's normalizer preserved.
297
+
298
+ ``traceback`` is present only for an episode or trace that actually failed
299
+ (spec section 6.12); a successful record carries none, because a stack
300
+ trace from a healthy run is host noise rather than evidence.
301
+ """
302
+
303
+ type: NonEmptyString
304
+ message: str
305
+ traceback: str | None = None
306
+
307
+
308
+ class NormalizedReward(ProtocolModel):
309
+ """One reward as Verifiers scored it, with its weighted contribution."""
310
+
311
+ name: NonEmptyString
312
+ score: float
313
+ weight: float
314
+ value: float
315
+
316
+ @model_validator(mode="after")
317
+ def _check_every_number_is_finite(self) -> Self:
318
+ """Reject a reward carrying a non-finite score, weight, or value."""
319
+ for field, number in (
320
+ ("score", self.score),
321
+ ("weight", self.weight),
322
+ ("value", self.value),
323
+ ):
324
+ if number != number or number in (float("inf"), float("-inf")):
325
+ raise ValueError(f"reward {field} must be finite; got {number!r}")
326
+ return self
327
+
328
+
329
+ class NormalizedUsage(ProtocolModel):
330
+ """Token consumption for one trace, and what the provider said it cost.
331
+
332
+ ``cost_usd`` is the provider's own figure and is absent whenever the
333
+ provider publishes none. Nothing here computes a cost from a price list:
334
+ an operational record downstream states where a cost came from, and a
335
+ number invented at this level could not be told apart from a reported one.
336
+ """
337
+
338
+ input_tokens: int = Field(ge=0)
339
+ output_tokens: int = Field(ge=0)
340
+ total_tokens: int = Field(ge=0)
341
+ cached_input_tokens: int | None = Field(default=None, ge=0)
342
+ cost_usd: float | None = Field(default=None, ge=0.0)
343
+
344
+
345
+ class NormalizedTool(ProtocolModel):
346
+ """One tool the subject was offered, described by digest.
347
+
348
+ A tool's description and parameter schema are prompt material. They are
349
+ hashed rather than copied so that a receipt can prove two variants were
350
+ offered identical tools without republishing the prompt surface.
351
+ """
352
+
353
+ name: NonEmptyString
354
+ description_digest: Digest
355
+ parameters_digest: Digest
356
+
357
+
358
+ class NormalizedRuntime(ProtocolModel):
359
+ """The box one trace ran in, as the evaluation recorded it.
360
+
361
+ ``image_index_digest`` is the content the reference names — for a
362
+ multi-platform repository, the OCI image index. The pinned Verifiers build
363
+ records the reference it was asked to run and nothing about what the daemon
364
+ resolved it to, so this is a projection of the request rather than a report
365
+ from the daemon; what the daemon confirmed is captured separately by
366
+ :class:`SubjectImageResolution` and the two are required to agree.
367
+ """
368
+
369
+ kind: Literal["docker"]
370
+ runtime_id: str | None
371
+ image: NonEmptyString
372
+ image_index_digest: Digest
373
+ cpu: float | None
374
+ memory_gb: float | None
375
+
376
+
377
+ class NormalizedTrace(ProtocolModel):
378
+ """One subject rollout, projected onto the fields a receipt may cite."""
379
+
380
+ trace_id: NonEmptyString
381
+ agent_role: Literal["subject"]
382
+ task_hash: Digest
383
+ ok: bool
384
+ # Recorded by the run itself rather than inferred: every upstream trace
385
+ # carries the Verifiers build that wrote it, which is what lets the pin be
386
+ # checked from the evidence instead of from a caller's claim about it.
387
+ verifiers_version: NonEmptyString
388
+ verifiers_revision: NonEmptyString
389
+ model_id: NonEmptyString
390
+ # The settings this rollout was actually sampled under, resolved by the
391
+ # engine and carried by the rollout itself. Two variants that disagree here
392
+ # were sampled differently, whatever their manifests declared.
393
+ sampling: dict[str, JsonValue]
394
+ harness_id: NonEmptyString
395
+ harness_version: NonEmptyString
396
+ use_bundled_skill: bool
397
+ skill_root_digests: list[Digest]
398
+ runtime: NormalizedRuntime
399
+ tools: list[NormalizedTool]
400
+ rewards: list[NormalizedReward]
401
+ metrics: dict[str, float | None]
402
+ usage: NormalizedUsage | None
403
+ model_calls: int = Field(ge=0)
404
+ num_turns: int = Field(ge=0)
405
+ last_reply: str | None
406
+ errors: list[NormalizedExecutionError]
407
+ raw_trace_digest: Digest
408
+
409
+ @model_validator(mode="after")
410
+ def _check_rewards_are_named_once(self) -> Self:
411
+ """Reject a trace that scores the same reward twice."""
412
+ names = [reward.name for reward in self.rewards]
413
+ if len(set(names)) != len(names):
414
+ raise ValueError("a trace records each reward exactly once")
415
+ return self
416
+
417
+ @model_validator(mode="after")
418
+ def _check_sampling_was_resolved(self) -> Self:
419
+ """Reject a trace that records no sampling settings at all."""
420
+ if not self.sampling:
421
+ raise ValueError(
422
+ "a trace records the sampling settings its rollout resolved"
423
+ )
424
+ return self
425
+
426
+ def reward(self, name: str) -> NormalizedReward | None:
427
+ """Return one reward by name, or ``None`` when it was not scored."""
428
+ for reward in self.rewards:
429
+ if reward.name == name:
430
+ return reward
431
+ return None
432
+
433
+
434
+ class NormalizedEpisode(ProtocolModel):
435
+ """One task's episode, ordered by the Campaign's committed membership."""
436
+
437
+ episode_id: NonEmptyString
438
+ env_id: NonEmptyString
439
+ task_hash: Digest
440
+ task_position: int = Field(ge=0)
441
+ ok: bool
442
+ traces: list[NormalizedTrace]
443
+ errors: list[NormalizedExecutionError]
444
+ raw_episode_digest: Digest
445
+
446
+ @model_validator(mode="after")
447
+ def _check_every_trace_belongs_to_this_task(self) -> Self:
448
+ """Reject an episode whose traces score a different task."""
449
+ for trace in self.traces:
450
+ if trace.task_hash != self.task_hash:
451
+ raise ValueError(
452
+ "an episode's traces all score the episode's own task; got "
453
+ f"{trace.task_hash} inside {self.task_hash}"
454
+ )
455
+ return self
456
+
457
+
458
+ # ---------------------------------------------------------------------------
459
+ # Results
460
+ # ---------------------------------------------------------------------------
461
+
462
+
463
+ class VariantExecutionResult(ProtocolModel):
464
+ """One variant, executed, with raw evidence and its normalized projection."""
465
+
466
+ variant: VariantName
467
+ experiment_manifest_digest: Digest
468
+ resolved_verifiers_config: ArtifactRef
469
+ raw_traces: ArtifactRef
470
+ eval_log: ArtifactRef
471
+ normalized_episodes: ArtifactRef
472
+ child_outcome: ChildProcessOutcome
473
+ image_resolution: SubjectImageResolution
474
+ episodes: list[NormalizedEpisode]
475
+
476
+ @model_validator(mode="after")
477
+ def _check_the_outcome_describes_this_variant(self) -> Self:
478
+ """Reject a result whose child outcome belongs to the other variant."""
479
+ if self.child_outcome.variant is not self.variant:
480
+ raise ValueError(
481
+ f"a {self.variant.value} result carries a "
482
+ f"{self.child_outcome.variant.value} child outcome"
483
+ )
484
+ if self.image_resolution.variant is not self.variant:
485
+ raise ValueError(
486
+ f"a {self.variant.value} result carries a "
487
+ f"{self.image_resolution.variant.value} image resolution"
488
+ )
489
+ return self
490
+
491
+
492
+ class RealExecutionResult(ProtocolModel):
493
+ """What WP6 hands WP7: both variants, executed under one schedule."""
494
+
495
+ execution_backend: Literal["verifiers"]
496
+ engine_digest: Digest
497
+ verifiers_revision: NonEmptyString
498
+ schedule: VariantSchedule
499
+ baseline: VariantExecutionResult
500
+ candidate: VariantExecutionResult
501
+
502
+ @model_validator(mode="after")
503
+ def _check_each_side_is_the_side_it_claims(self) -> Self:
504
+ """Reject a result that files a variant under the wrong name."""
505
+ if self.baseline.variant is not VariantName.BASELINE:
506
+ raise ValueError("the baseline slot holds the baseline variant")
507
+ if self.candidate.variant is not VariantName.CANDIDATE:
508
+ raise ValueError("the candidate slot holds the candidate variant")
509
+ return self
510
+
511
+
512
+ # ---------------------------------------------------------------------------
513
+ # Checks
514
+ # ---------------------------------------------------------------------------
515
+
516
+
517
+ class ExecutionCheck(ProtocolModel):
518
+ """One named question about an execution, and its answer.
519
+
520
+ The same shape as ``ValidationCheck`` (spec section 21.5) and for the same
521
+ reason: a caller reads an ordered list of named verdicts rather than
522
+ catching exceptions to discover which rule failed.
523
+ """
524
+
525
+ id: NonEmptyString
526
+ status: Literal["passed", "failed", "warning", "not_run"]
527
+ detail: NonEmptyString