techtree 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- techtree/__init__.py +35 -0
- techtree/__main__.py +14 -0
- techtree/canonical.py +239 -0
- techtree/catalog/__init__.py +25 -0
- techtree/catalog/repository.py +400 -0
- techtree/catalog/service.py +419 -0
- techtree/cli/__init__.py +1 -0
- techtree/cli/app.py +416 -0
- techtree/cli/commands/__init__.py +1 -0
- techtree/cli/commands/climb.py +1223 -0
- techtree/cli/commands/doctor.py +147 -0
- techtree/cli/commands/engine.py +207 -0
- techtree/cli/commands/proof.py +556 -0
- techtree/cli/commands/publish.py +447 -0
- techtree/cli/commands/release.py +303 -0
- techtree/cli/commands/run.py +1067 -0
- techtree/cli/commands/setup.py +181 -0
- techtree/cli/commands/skill.py +221 -0
- techtree/cli/commands/uplift.py +698 -0
- techtree/cli/commands/withdraw.py +212 -0
- techtree/cli/confirm.py +47 -0
- techtree/cli/context.py +96 -0
- techtree/cli/invoke.py +220 -0
- techtree/cli/output.py +280 -0
- techtree/constants.py +138 -0
- techtree/crypto.py +128 -0
- techtree/doctor/__init__.py +1 -0
- techtree/doctor/checks.py +675 -0
- techtree/doctor/execution_checks.py +435 -0
- techtree/doctor/service.py +326 -0
- techtree/drafts/__init__.py +32 -0
- techtree/drafts/source.py +146 -0
- techtree/drafts/store.py +992 -0
- techtree/engines/__init__.py +1 -0
- techtree/engines/bundle.py +251 -0
- techtree/engines/installer.py +679 -0
- techtree/engines/registry.py +235 -0
- techtree/engines/runner.py +170 -0
- techtree/errors.py +262 -0
- techtree/fs.py +234 -0
- techtree/harness.py +108 -0
- techtree/identity/__init__.py +41 -0
- techtree/identity/models.py +113 -0
- techtree/identity/service.py +199 -0
- techtree/identity/store.py +263 -0
- techtree/ids.py +85 -0
- techtree/manifests/__init__.py +39 -0
- techtree/manifests/builder.py +433 -0
- techtree/manifests/compare.py +376 -0
- techtree/models/__init__.py +282 -0
- techtree/models/base.py +201 -0
- techtree/models/campaign.py +484 -0
- techtree/models/catalog.py +227 -0
- techtree/models/cli.py +151 -0
- techtree/models/climb.py +254 -0
- techtree/models/data_policy.py +130 -0
- techtree/models/engine.py +156 -0
- techtree/models/episode_receipt.py +130 -0
- techtree/models/evaluation_backend.py +113 -0
- techtree/models/experiment.py +154 -0
- techtree/models/run.py +214 -0
- techtree/models/skill.py +156 -0
- techtree/models/uplift_report.py +158 -0
- techtree/models/validation.py +299 -0
- techtree/paths.py +116 -0
- techtree/presentation/__init__.py +31 -0
- techtree/presentation/build.py +1242 -0
- techtree/presentation/compact.py +246 -0
- techtree/presentation/evidence.py +169 -0
- techtree/presentation/models.py +358 -0
- techtree/presentation/rich.py +312 -0
- techtree/presentation/sanitize.py +156 -0
- techtree/publication/__init__.py +44 -0
- techtree/publication/address.py +180 -0
- techtree/publication/coordinates.py +26 -0
- techtree/publication/journal.py +212 -0
- techtree/publication/keccak.py +183 -0
- techtree/publication/models.py +209 -0
- techtree/publication/offer.py +35 -0
- techtree/publication/service.py +618 -0
- techtree/publication/transport.py +296 -0
- techtree/publication/verify.py +242 -0
- techtree/publication/withdraw.py +156 -0
- techtree/py.typed +0 -0
- techtree/receipts/__init__.py +52 -0
- techtree/receipts/bundle.py +578 -0
- techtree/receipts/compare.py +1065 -0
- techtree/receipts/episode.py +672 -0
- techtree/receipts/execution.py +630 -0
- techtree/receipts/observed.py +474 -0
- techtree/receipts/set.py +336 -0
- techtree/receipts/uplift.py +655 -0
- techtree/receipts/verify.py +1055 -0
- techtree/release/__init__.py +9 -0
- techtree/release/bootstrap.py +509 -0
- techtree/release/checks.py +376 -0
- techtree/release/document.py +125 -0
- techtree/release/generate.py +221 -0
- techtree/release/models.py +293 -0
- techtree/release/provenance.py +109 -0
- techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
- techtree/resources/catalog/catalog.json +32 -0
- techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
- techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
- techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
- techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
- techtree/resources/engines/default/engine.json +20 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
- techtree/resources/engines/default/pyproject.toml +23 -0
- techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
- techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
- techtree/resources/engines/default/tools/normalize_validation.py +222 -0
- techtree/resources/engines/default/uv.lock +1758 -0
- techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
- techtree/resources/release/build-provenance.json +4 -0
- techtree/resources/release/release-core.json +24 -0
- techtree/runs/__init__.py +31 -0
- techtree/runs/artifacts.py +750 -0
- techtree/runs/child_registry.py +228 -0
- techtree/runs/events.py +478 -0
- techtree/runs/executor.py +140 -0
- techtree/runs/fake.py +741 -0
- techtree/runs/launcher.py +253 -0
- techtree/runs/machine.py +489 -0
- techtree/runs/real.py +789 -0
- techtree/runs/service.py +616 -0
- techtree/runs/store.py +555 -0
- techtree/runs/validation.py +259 -0
- techtree/runs/variants.py +684 -0
- techtree/settings.py +143 -0
- techtree/skills/__init__.py +14 -0
- techtree/skills/archive.py +282 -0
- techtree/skills/policy.py +62 -0
- techtree/skills/scanner.py +394 -0
- techtree/skills/service.py +752 -0
- techtree/skills/starter.py +434 -0
- techtree/tasksets/__init__.py +1 -0
- techtree/tasksets/membership.py +269 -0
- techtree/tasksets/provider.py +207 -0
- techtree/tasksets/resolver.py +311 -0
- techtree/tasksets/service.py +484 -0
- techtree/tasksets/verifiers_cli.py +538 -0
- techtree/uplift/__init__.py +20 -0
- techtree/uplift/context.py +544 -0
- techtree/uplift/derive.py +203 -0
- techtree/uplift/public_tasks.py +151 -0
- techtree/uplift/service.py +719 -0
- techtree/uplift/source.py +160 -0
- techtree/verifiers/__init__.py +31 -0
- techtree/verifiers/budget.py +219 -0
- techtree/verifiers/child.py +633 -0
- techtree/verifiers/compiler.py +432 -0
- techtree/verifiers/config.py +365 -0
- techtree/verifiers/credentials.py +321 -0
- techtree/verifiers/image.py +126 -0
- techtree/verifiers/models.py +527 -0
- techtree/verifiers/outputs.py +368 -0
- techtree/verifiers/progress.py +192 -0
- techtree/verifiers/supervisor.py +341 -0
- techtree/verifiers/verify.py +782 -0
- techtree/version.py +39 -0
- techtree/worker/__init__.py +18 -0
- techtree/worker/execute.py +487 -0
- techtree/worker/main.py +57 -0
- techtree-0.1.0.dist-info/METADATA +344 -0
- techtree-0.1.0.dist-info/RECORD +174 -0
- techtree-0.1.0.dist-info/WHEEL +4 -0
- techtree-0.1.0.dist-info/entry_points.txt +3 -0
- techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""Asking the local daemon what the pinned subject image is. Decisions 0007 R5.
|
|
2
|
+
|
|
3
|
+
The evaluation cannot answer this. The pinned Verifiers build hands Docker an
|
|
4
|
+
image reference and records the reference, so its own output says what was
|
|
5
|
+
*asked for* and nothing about what was *there*. A comparison built on that alone
|
|
6
|
+
can only report the Campaign's own pin back to the reader, which is why the
|
|
7
|
+
image digest used to be a warning rather than a check.
|
|
8
|
+
|
|
9
|
+
So Techtree asks, once per variant, immediately before that variant's child is
|
|
10
|
+
launched. Two facts come back, both from the daemon:
|
|
11
|
+
|
|
12
|
+
*the content it holds* — ``docker image inspect`` resolves the pinned reference
|
|
13
|
+
or fails. A daemon that resolves ``repository@sha256:...`` is holding exactly
|
|
14
|
+
that content, and the repository digests it lists for the image are required to
|
|
15
|
+
include the pin, so a resolution that came from somewhere else is refused;
|
|
16
|
+
|
|
17
|
+
*the platform it serves* — the operating system and architecture the index was
|
|
18
|
+
resolved to on this host. The platform-specific manifest digest is *not* asked
|
|
19
|
+
of the daemon, because the daemon does not know it: an image pulled by index
|
|
20
|
+
digest keeps the index digest and the unpacked platform image, not the platform
|
|
21
|
+
manifest's own digest. That digest is a property of the pinned index, recorded
|
|
22
|
+
per platform in the Campaign, and read out by the platform observed here.
|
|
23
|
+
|
|
24
|
+
Nothing here pulls. Provisioning an image is an explicit setup step the operator
|
|
25
|
+
asks for (spec section 6.18), and an executor that downloaded a few hundred
|
|
26
|
+
megabytes to make its own check pass would be spending somebody's bandwidth to
|
|
27
|
+
avoid telling them the truth.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
import json
|
|
33
|
+
import shutil
|
|
34
|
+
import subprocess
|
|
35
|
+
from collections.abc import Iterable
|
|
36
|
+
from typing import Final
|
|
37
|
+
|
|
38
|
+
from techtree.errors import ValidationError
|
|
39
|
+
from techtree.models.base import JsonValue
|
|
40
|
+
from techtree.models.campaign import RuntimeSpec
|
|
41
|
+
from techtree.verifiers.models import SubjectImageResolution, VariantName
|
|
42
|
+
|
|
43
|
+
__all__ = [
|
|
44
|
+
"IMAGE_INSPECT_TIMEOUT_SECONDS",
|
|
45
|
+
"SUBJECT_IMAGE_UNRESOLVED",
|
|
46
|
+
"resolve_subject_image",
|
|
47
|
+
]
|
|
48
|
+
|
|
49
|
+
#: Stable error code. Spec section 15.
|
|
50
|
+
SUBJECT_IMAGE_UNRESOLVED: Final = "subject_image_unresolved"
|
|
51
|
+
|
|
52
|
+
#: Inspecting a local image is a metadata read, but the daemon may still be
|
|
53
|
+
#: waking up.
|
|
54
|
+
IMAGE_INSPECT_TIMEOUT_SECONDS: Final = 30.0
|
|
55
|
+
|
|
56
|
+
_INSPECT_FORMAT: Final = "{{json .RepoDigests}}\t{{.Os}}/{{.Architecture}}"
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def resolve_subject_image(
|
|
60
|
+
runtime: RuntimeSpec, variant: VariantName
|
|
61
|
+
) -> SubjectImageResolution:
|
|
62
|
+
"""Return what this machine's daemon holds for the Campaign's subject image."""
|
|
63
|
+
if shutil.which("docker") is None:
|
|
64
|
+
raise ValidationError(
|
|
65
|
+
"docker is not on PATH, so what the subject container would run "
|
|
66
|
+
"cannot be established",
|
|
67
|
+
code=SUBJECT_IMAGE_UNRESOLVED,
|
|
68
|
+
details={"variant": variant.value, "image": runtime.image},
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
completed = subprocess.run(
|
|
72
|
+
["docker", "image", "inspect", runtime.image, "--format", _INSPECT_FORMAT],
|
|
73
|
+
capture_output=True,
|
|
74
|
+
text=True,
|
|
75
|
+
check=False,
|
|
76
|
+
timeout=IMAGE_INSPECT_TIMEOUT_SECONDS,
|
|
77
|
+
stdin=subprocess.DEVNULL,
|
|
78
|
+
)
|
|
79
|
+
if completed.returncode != 0 or not completed.stdout.strip():
|
|
80
|
+
raise ValidationError(
|
|
81
|
+
f"the Docker daemon does not hold {runtime.image}; pull it as an "
|
|
82
|
+
"explicit setup step before running an evaluation",
|
|
83
|
+
code=SUBJECT_IMAGE_UNRESOLVED,
|
|
84
|
+
details={
|
|
85
|
+
"variant": variant.value,
|
|
86
|
+
"image": runtime.image,
|
|
87
|
+
"exit_code": completed.returncode,
|
|
88
|
+
},
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
digests, _, platform = completed.stdout.strip().partition("\t")
|
|
92
|
+
repository_digests = json.loads(digests)
|
|
93
|
+
if runtime.image not in repository_digests:
|
|
94
|
+
raise ValidationError(
|
|
95
|
+
"the image the daemon resolved does not list the content the "
|
|
96
|
+
"Campaign pinned among its own repository digests",
|
|
97
|
+
code=SUBJECT_IMAGE_UNRESOLVED,
|
|
98
|
+
details={
|
|
99
|
+
"variant": variant.value,
|
|
100
|
+
"image": runtime.image,
|
|
101
|
+
"repository_digests": _text_detail(repository_digests),
|
|
102
|
+
},
|
|
103
|
+
)
|
|
104
|
+
if platform not in runtime.image_platform_digests:
|
|
105
|
+
raise ValidationError(
|
|
106
|
+
f"the daemon serves {runtime.image} as {platform}, which the "
|
|
107
|
+
"Campaign pins no manifest digest for",
|
|
108
|
+
code=SUBJECT_IMAGE_UNRESOLVED,
|
|
109
|
+
details={
|
|
110
|
+
"variant": variant.value,
|
|
111
|
+
"platform": platform,
|
|
112
|
+
"pinned_platforms": _text_detail(runtime.image_platform_digests),
|
|
113
|
+
},
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
return SubjectImageResolution(
|
|
117
|
+
variant=variant,
|
|
118
|
+
image=runtime.image,
|
|
119
|
+
index_digest=runtime.image_index_digest,
|
|
120
|
+
platform=platform,
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _text_detail(values: Iterable[object]) -> list[JsonValue]:
|
|
125
|
+
"""Return an ordered, printable list in the shape error details carry."""
|
|
126
|
+
return [text for text in sorted(str(value) for value in values)]
|
|
@@ -0,0 +1,527 @@
|
|
|
1
|
+
"""Local integration types for native Verifiers execution. Spec section 6.6.
|
|
2
|
+
|
|
3
|
+
Nothing in this module is a Techtree protocol root. These objects describe one
|
|
4
|
+
machine's execution of one Campaign: which child ran, where its bytes landed,
|
|
5
|
+
and what the pinned engine's normalizer made of them. They are hashed and
|
|
6
|
+
written into a run directory, never into the Campaign graph and never onto a
|
|
7
|
+
website.
|
|
8
|
+
|
|
9
|
+
``RunPaths`` is the one addition the specification names but the repository did
|
|
10
|
+
not yet have (spec section 6.19). It exists here rather than in
|
|
11
|
+
``techtree.paths`` because every path it owns belongs to Verifiers execution;
|
|
12
|
+
``TechtreePaths`` stays the answer to "where does Techtree keep its state", and
|
|
13
|
+
this stays the answer to "where does one variant's evaluation live".
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
from datetime import datetime
|
|
20
|
+
from enum import StrEnum
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import Final, Literal, Self
|
|
23
|
+
|
|
24
|
+
from pydantic import Field, model_validator
|
|
25
|
+
|
|
26
|
+
from techtree.models.base import (
|
|
27
|
+
ArtifactRef,
|
|
28
|
+
Digest,
|
|
29
|
+
JsonValue,
|
|
30
|
+
NonEmptyString,
|
|
31
|
+
ProtocolModel,
|
|
32
|
+
)
|
|
33
|
+
from techtree.models.campaign import VariantSchedule
|
|
34
|
+
from techtree.paths import TechtreePaths
|
|
35
|
+
|
|
36
|
+
__all__ = [
|
|
37
|
+
"COMMAND_LOG_FILENAME",
|
|
38
|
+
"EVAL_RUN_NAME",
|
|
39
|
+
"INPUT_CONFIG_FILENAME",
|
|
40
|
+
"NORMALIZED_EPISODES_FILENAME",
|
|
41
|
+
"STDERR_LOG_FILENAME",
|
|
42
|
+
"STDOUT_LOG_FILENAME",
|
|
43
|
+
"SUPERVISION_RECORD_FILENAME",
|
|
44
|
+
"VERIFIERS_DIRECTORY",
|
|
45
|
+
"ChildProcessOutcome",
|
|
46
|
+
"ExecutionCheck",
|
|
47
|
+
"NormalizedEpisode",
|
|
48
|
+
"NormalizedExecutionError",
|
|
49
|
+
"NormalizedReward",
|
|
50
|
+
"NormalizedRuntime",
|
|
51
|
+
"NormalizedTool",
|
|
52
|
+
"NormalizedTrace",
|
|
53
|
+
"NormalizedUsage",
|
|
54
|
+
"RealExecutionResult",
|
|
55
|
+
"RunPaths",
|
|
56
|
+
"SubjectImageResolution",
|
|
57
|
+
"VariantExecutionPlan",
|
|
58
|
+
"VariantExecutionResult",
|
|
59
|
+
"VariantName",
|
|
60
|
+
]
|
|
61
|
+
|
|
62
|
+
#: The run-owned subtree every variant's evaluation lives under.
|
|
63
|
+
VERIFIERS_DIRECTORY: Final = "verifiers"
|
|
64
|
+
#: The configuration Techtree compiles, as opposed to the one the engine
|
|
65
|
+
#: resolves and writes back out under its own name. The extension is load
|
|
66
|
+
#: bearing rather than cosmetic: the engine chooses its parser from it, and
|
|
67
|
+
#: JSON is the only one of the two formats that can spell the explicit null
|
|
68
|
+
#: that turns the live dashboard off (``src/techtree/verifiers/config.py``).
|
|
69
|
+
INPUT_CONFIG_FILENAME: Final = "input.json"
|
|
70
|
+
|
|
71
|
+
#: What the engine writes when it normalizes one variant's raw episodes. It sits
|
|
72
|
+
#: beside the raw evidence inside ``run/`` (spec section 6.19) rather than above
|
|
73
|
+
#: it, so that one directory holds everything one execution produced.
|
|
74
|
+
NORMALIZED_EPISODES_FILENAME: Final = "normalized-episodes.jsonl"
|
|
75
|
+
|
|
76
|
+
#: Where a live child's captured streams land. Spec section 6.10 redirects both
|
|
77
|
+
#: to run-owned files; section 6.19 names them.
|
|
78
|
+
STDOUT_LOG_FILENAME: Final = "stdout.log"
|
|
79
|
+
STDERR_LOG_FILENAME: Final = "stderr.log"
|
|
80
|
+
|
|
81
|
+
#: The dry run is short and captured whole, so it is recorded in one file rather
|
|
82
|
+
#: than as a pair of streams. Spec section 6.19.
|
|
83
|
+
COMMAND_LOG_FILENAME: Final = "command.log"
|
|
84
|
+
|
|
85
|
+
#: What one variant's supervisor leaves behind: why the evaluation ended, and
|
|
86
|
+
#: how long it took to stop (decisions document 0029, layer B). It sits beside
|
|
87
|
+
#: the evaluation rather than inside ``run/`` because the engine owns that
|
|
88
|
+
#: directory and this is Techtree's own record of the engine's lifetime.
|
|
89
|
+
SUPERVISION_RECORD_FILENAME: Final = "supervision.json"
|
|
90
|
+
|
|
91
|
+
#: What Techtree calls the one evaluation it groups under a variant's output
|
|
92
|
+
#: directory. Since v0.3.1 ``--output-dir`` names the directory runs are
|
|
93
|
+
#: *grouped* under and the run itself lands in ``<output-dir>/<run.dir>``, with
|
|
94
|
+
#: a random suffix when nothing names it (``docs/verifiers-pin-0.3.1.md``,
|
|
95
|
+
#: deviation D2). Naming it is what makes one variant's evidence findable, and
|
|
96
|
+
#: findable twice.
|
|
97
|
+
EVAL_RUN_NAME: Final = "run"
|
|
98
|
+
|
|
99
|
+
_DRY_RUN_DIRECTORY: Final = "dry-run"
|
|
100
|
+
_INPUTS_DIRECTORY: Final = "inputs"
|
|
101
|
+
_MANIFESTS_DIRECTORY: Final = "manifests"
|
|
102
|
+
_SKILL_FILES_PATH: Final = ("skill", "files")
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
class VariantName(StrEnum):
|
|
106
|
+
"""Which side of the comparison a child process is running."""
|
|
107
|
+
|
|
108
|
+
BASELINE = "baseline"
|
|
109
|
+
CANDIDATE = "candidate"
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
# ---------------------------------------------------------------------------
|
|
113
|
+
# Where one run's evaluation lives
|
|
114
|
+
# ---------------------------------------------------------------------------
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
@dataclass(frozen=True)
|
|
118
|
+
class RunPaths:
|
|
119
|
+
"""Every path one run's Verifiers execution is allowed to touch.
|
|
120
|
+
|
|
121
|
+
Spec section 6.19. The layout is per variant so that a parallel schedule
|
|
122
|
+
cannot have two children writing the same file, and so that a cancelled
|
|
123
|
+
variant's partial evidence stays legible next to its sibling's complete
|
|
124
|
+
evidence.
|
|
125
|
+
"""
|
|
126
|
+
|
|
127
|
+
root: Path
|
|
128
|
+
|
|
129
|
+
@classmethod
|
|
130
|
+
def for_run(cls, paths: TechtreePaths, run_id: str) -> Self:
|
|
131
|
+
"""Locate one run's directory inside a Techtree home."""
|
|
132
|
+
return cls(root=paths.run_dir(run_id))
|
|
133
|
+
|
|
134
|
+
@property
|
|
135
|
+
def inputs_dir(self) -> Path:
|
|
136
|
+
"""The run's own copies of everything it executes."""
|
|
137
|
+
return self.root / _INPUTS_DIRECTORY
|
|
138
|
+
|
|
139
|
+
@property
|
|
140
|
+
def skill_files_dir(self) -> Path:
|
|
141
|
+
"""The run-owned skill tree a compiled config may point at."""
|
|
142
|
+
return self.inputs_dir.joinpath(*_SKILL_FILES_PATH)
|
|
143
|
+
|
|
144
|
+
def manifest_path(self, variant: VariantName) -> Path:
|
|
145
|
+
"""The run's own copy of one variant's experiment manifest."""
|
|
146
|
+
return self.inputs_dir / _MANIFESTS_DIRECTORY / f"{variant.value}.json"
|
|
147
|
+
|
|
148
|
+
@property
|
|
149
|
+
def verifiers_dir(self) -> Path:
|
|
150
|
+
"""The root of every variant's evaluation."""
|
|
151
|
+
return self.root / VERIFIERS_DIRECTORY
|
|
152
|
+
|
|
153
|
+
def variant_dir(self, variant: VariantName) -> Path:
|
|
154
|
+
"""One variant's evaluation directory."""
|
|
155
|
+
return self.verifiers_dir / variant.value
|
|
156
|
+
|
|
157
|
+
def variant_input_config(self, variant: VariantName) -> Path:
|
|
158
|
+
"""The configuration Techtree compiles for one variant."""
|
|
159
|
+
return self.variant_dir(variant) / INPUT_CONFIG_FILENAME
|
|
160
|
+
|
|
161
|
+
def variant_dry_run_dir(self, variant: VariantName) -> Path:
|
|
162
|
+
"""Where the engine writes the resolved config during validation.
|
|
163
|
+
|
|
164
|
+
Separate from the run directory on purpose: a dry run writes only the
|
|
165
|
+
resolved configuration (``docs/verifiers-eval.md``, finding E2), and
|
|
166
|
+
letting it land beside real evidence would leave a directory that looks
|
|
167
|
+
like a truncated run.
|
|
168
|
+
"""
|
|
169
|
+
return self.variant_dir(variant) / _DRY_RUN_DIRECTORY
|
|
170
|
+
|
|
171
|
+
def variant_dry_run_command_log(self, variant: VariantName) -> Path:
|
|
172
|
+
"""What the dry-run invocation was, and what it said back."""
|
|
173
|
+
return self.variant_dry_run_dir(variant) / COMMAND_LOG_FILENAME
|
|
174
|
+
|
|
175
|
+
def variant_supervision_record(self, variant: VariantName) -> Path:
|
|
176
|
+
"""Where one variant's supervisor records how its evaluation ended."""
|
|
177
|
+
return self.variant_dir(variant) / SUPERVISION_RECORD_FILENAME
|
|
178
|
+
|
|
179
|
+
def variant_output_group_dir(self, variant: VariantName) -> Path:
|
|
180
|
+
"""The directory the engine groups one variant's evaluations under.
|
|
181
|
+
|
|
182
|
+
This is what the compiled configuration's ``output_dir`` names, and it
|
|
183
|
+
is a level above the run itself: ``--output-dir`` groups runs rather
|
|
184
|
+
than receiving one (deviation D2).
|
|
185
|
+
"""
|
|
186
|
+
return self.variant_dir(variant)
|
|
187
|
+
|
|
188
|
+
def variant_output_dir(self, variant: VariantName) -> Path:
|
|
189
|
+
"""Where the engine writes one variant's real evaluation output.
|
|
190
|
+
|
|
191
|
+
One level below the group directory, under the name the invocation
|
|
192
|
+
pins with ``--run.name``. Left unpinned the engine would append a
|
|
193
|
+
random suffix here and the evidence would land somewhere Techtree
|
|
194
|
+
never looks.
|
|
195
|
+
"""
|
|
196
|
+
return self.variant_output_group_dir(variant) / EVAL_RUN_NAME
|
|
197
|
+
|
|
198
|
+
def variant_stdout_log(self, variant: VariantName) -> Path:
|
|
199
|
+
"""Where one variant's child sends everything it prints.
|
|
200
|
+
|
|
201
|
+
Never a console. With ``rich`` disabled the pinned CLI dumps every
|
|
202
|
+
trace as indented JSON when the run ends, and those are the subject's
|
|
203
|
+
transcripts (``docs/verifiers-eval.md``).
|
|
204
|
+
"""
|
|
205
|
+
return self.variant_output_dir(variant) / STDOUT_LOG_FILENAME
|
|
206
|
+
|
|
207
|
+
def variant_stderr_log(self, variant: VariantName) -> Path:
|
|
208
|
+
"""Where one variant's child sends its diagnostics."""
|
|
209
|
+
return self.variant_output_dir(variant) / STDERR_LOG_FILENAME
|
|
210
|
+
|
|
211
|
+
def variant_normalized_episodes(self, variant: VariantName) -> Path:
|
|
212
|
+
"""One variant's normalized projection, beside the evidence it projects."""
|
|
213
|
+
return self.variant_output_dir(variant) / NORMALIZED_EPISODES_FILENAME
|
|
214
|
+
|
|
215
|
+
def relative(self, path: Path) -> str:
|
|
216
|
+
"""Return ``path`` as a POSIX path relative to the run directory."""
|
|
217
|
+
return path.relative_to(self.root).as_posix()
|
|
218
|
+
|
|
219
|
+
def owns(self, path: Path) -> bool:
|
|
220
|
+
"""Whether ``path`` lies inside the run's own input tree."""
|
|
221
|
+
return path.is_absolute() and path.is_relative_to(self.inputs_dir)
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
# ---------------------------------------------------------------------------
|
|
225
|
+
# Planning and process outcome
|
|
226
|
+
# ---------------------------------------------------------------------------
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
class VariantExecutionPlan(ProtocolModel):
|
|
230
|
+
"""Everything one variant's child process needs, resolved."""
|
|
231
|
+
|
|
232
|
+
variant: VariantName
|
|
233
|
+
experiment_manifest_digest: Digest
|
|
234
|
+
experiment_manifest_path: NonEmptyString
|
|
235
|
+
verifiers_input_config_path: NonEmptyString
|
|
236
|
+
verifiers_output_dir: NonEmptyString
|
|
237
|
+
skill_paths: list[NonEmptyString]
|
|
238
|
+
task_count: int = Field(ge=1)
|
|
239
|
+
max_concurrent: int = Field(ge=1)
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
class SubjectImageResolution(ProtocolModel):
|
|
243
|
+
"""What the local daemon answered about the pinned subject image.
|
|
244
|
+
|
|
245
|
+
The pinned Verifiers build asks Docker for a reference and records only the
|
|
246
|
+
reference, so the evaluation's own output cannot say what the daemon held or
|
|
247
|
+
which platform it served. Techtree asks, once per variant, immediately
|
|
248
|
+
before that variant's child is launched, and records the answer here.
|
|
249
|
+
|
|
250
|
+
Two facts, both from the daemon: the content digest it holds for the
|
|
251
|
+
reference — for a multi-platform repository, the OCI image index — and the
|
|
252
|
+
platform it resolved that index to on this host. The platform-specific
|
|
253
|
+
manifest digest is not asked of the daemon because the daemon does not know
|
|
254
|
+
it; it is a property of the pinned index, recorded in the Campaign per
|
|
255
|
+
supported platform, and the comparison reads it out by the platform observed
|
|
256
|
+
here.
|
|
257
|
+
"""
|
|
258
|
+
|
|
259
|
+
variant: VariantName
|
|
260
|
+
image: NonEmptyString
|
|
261
|
+
index_digest: Digest
|
|
262
|
+
platform: NonEmptyString
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
class ChildProcessOutcome(ProtocolModel):
|
|
266
|
+
"""What one Verifiers child process did.
|
|
267
|
+
|
|
268
|
+
``argv_digest`` rather than the argv itself. The compiled invocation never
|
|
269
|
+
carries a secret, but a digest is what makes that claim checkable without
|
|
270
|
+
re-reading a command line into a log.
|
|
271
|
+
"""
|
|
272
|
+
|
|
273
|
+
variant: VariantName
|
|
274
|
+
argv_digest: Digest
|
|
275
|
+
exit_code: int
|
|
276
|
+
started_at: datetime
|
|
277
|
+
finished_at: datetime
|
|
278
|
+
stdout_artifact: ArtifactRef
|
|
279
|
+
stderr_artifact: ArtifactRef
|
|
280
|
+
cancelled: bool
|
|
281
|
+
|
|
282
|
+
@model_validator(mode="after")
|
|
283
|
+
def _check_the_clock_moves_forward(self) -> Self:
|
|
284
|
+
"""Reject a process that finished before it started."""
|
|
285
|
+
if self.finished_at < self.started_at:
|
|
286
|
+
raise ValueError("a child process cannot finish before it starts")
|
|
287
|
+
return self
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
# ---------------------------------------------------------------------------
|
|
291
|
+
# The normalized projection of upstream evidence
|
|
292
|
+
# ---------------------------------------------------------------------------
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
class NormalizedExecutionError(ProtocolModel):
|
|
296
|
+
"""One failure the engine's normalizer preserved.
|
|
297
|
+
|
|
298
|
+
``traceback`` is present only for an episode or trace that actually failed
|
|
299
|
+
(spec section 6.12); a successful record carries none, because a stack
|
|
300
|
+
trace from a healthy run is host noise rather than evidence.
|
|
301
|
+
"""
|
|
302
|
+
|
|
303
|
+
type: NonEmptyString
|
|
304
|
+
message: str
|
|
305
|
+
traceback: str | None = None
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
class NormalizedReward(ProtocolModel):
|
|
309
|
+
"""One reward as Verifiers scored it, with its weighted contribution."""
|
|
310
|
+
|
|
311
|
+
name: NonEmptyString
|
|
312
|
+
score: float
|
|
313
|
+
weight: float
|
|
314
|
+
value: float
|
|
315
|
+
|
|
316
|
+
@model_validator(mode="after")
|
|
317
|
+
def _check_every_number_is_finite(self) -> Self:
|
|
318
|
+
"""Reject a reward carrying a non-finite score, weight, or value."""
|
|
319
|
+
for field, number in (
|
|
320
|
+
("score", self.score),
|
|
321
|
+
("weight", self.weight),
|
|
322
|
+
("value", self.value),
|
|
323
|
+
):
|
|
324
|
+
if number != number or number in (float("inf"), float("-inf")):
|
|
325
|
+
raise ValueError(f"reward {field} must be finite; got {number!r}")
|
|
326
|
+
return self
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
class NormalizedUsage(ProtocolModel):
|
|
330
|
+
"""Token consumption for one trace, and what the provider said it cost.
|
|
331
|
+
|
|
332
|
+
``cost_usd`` is the provider's own figure and is absent whenever the
|
|
333
|
+
provider publishes none. Nothing here computes a cost from a price list:
|
|
334
|
+
an operational record downstream states where a cost came from, and a
|
|
335
|
+
number invented at this level could not be told apart from a reported one.
|
|
336
|
+
"""
|
|
337
|
+
|
|
338
|
+
input_tokens: int = Field(ge=0)
|
|
339
|
+
output_tokens: int = Field(ge=0)
|
|
340
|
+
total_tokens: int = Field(ge=0)
|
|
341
|
+
cached_input_tokens: int | None = Field(default=None, ge=0)
|
|
342
|
+
cost_usd: float | None = Field(default=None, ge=0.0)
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
class NormalizedTool(ProtocolModel):
|
|
346
|
+
"""One tool the subject was offered, described by digest.
|
|
347
|
+
|
|
348
|
+
A tool's description and parameter schema are prompt material. They are
|
|
349
|
+
hashed rather than copied so that a receipt can prove two variants were
|
|
350
|
+
offered identical tools without republishing the prompt surface.
|
|
351
|
+
"""
|
|
352
|
+
|
|
353
|
+
name: NonEmptyString
|
|
354
|
+
description_digest: Digest
|
|
355
|
+
parameters_digest: Digest
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
class NormalizedRuntime(ProtocolModel):
|
|
359
|
+
"""The box one trace ran in, as the evaluation recorded it.
|
|
360
|
+
|
|
361
|
+
``image_index_digest`` is the content the reference names — for a
|
|
362
|
+
multi-platform repository, the OCI image index. The pinned Verifiers build
|
|
363
|
+
records the reference it was asked to run and nothing about what the daemon
|
|
364
|
+
resolved it to, so this is a projection of the request rather than a report
|
|
365
|
+
from the daemon; what the daemon confirmed is captured separately by
|
|
366
|
+
:class:`SubjectImageResolution` and the two are required to agree.
|
|
367
|
+
"""
|
|
368
|
+
|
|
369
|
+
kind: Literal["docker"]
|
|
370
|
+
runtime_id: str | None
|
|
371
|
+
image: NonEmptyString
|
|
372
|
+
image_index_digest: Digest
|
|
373
|
+
cpu: float | None
|
|
374
|
+
memory_gb: float | None
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
class NormalizedTrace(ProtocolModel):
|
|
378
|
+
"""One subject rollout, projected onto the fields a receipt may cite."""
|
|
379
|
+
|
|
380
|
+
trace_id: NonEmptyString
|
|
381
|
+
agent_role: Literal["subject"]
|
|
382
|
+
task_hash: Digest
|
|
383
|
+
ok: bool
|
|
384
|
+
# Recorded by the run itself rather than inferred: every upstream trace
|
|
385
|
+
# carries the Verifiers build that wrote it, which is what lets the pin be
|
|
386
|
+
# checked from the evidence instead of from a caller's claim about it.
|
|
387
|
+
verifiers_version: NonEmptyString
|
|
388
|
+
verifiers_revision: NonEmptyString
|
|
389
|
+
model_id: NonEmptyString
|
|
390
|
+
# The settings this rollout was actually sampled under, resolved by the
|
|
391
|
+
# engine and carried by the rollout itself. Two variants that disagree here
|
|
392
|
+
# were sampled differently, whatever their manifests declared.
|
|
393
|
+
sampling: dict[str, JsonValue]
|
|
394
|
+
harness_id: NonEmptyString
|
|
395
|
+
harness_version: NonEmptyString
|
|
396
|
+
use_bundled_skill: bool
|
|
397
|
+
skill_root_digests: list[Digest]
|
|
398
|
+
runtime: NormalizedRuntime
|
|
399
|
+
tools: list[NormalizedTool]
|
|
400
|
+
rewards: list[NormalizedReward]
|
|
401
|
+
metrics: dict[str, float | None]
|
|
402
|
+
usage: NormalizedUsage | None
|
|
403
|
+
model_calls: int = Field(ge=0)
|
|
404
|
+
num_turns: int = Field(ge=0)
|
|
405
|
+
last_reply: str | None
|
|
406
|
+
errors: list[NormalizedExecutionError]
|
|
407
|
+
raw_trace_digest: Digest
|
|
408
|
+
|
|
409
|
+
@model_validator(mode="after")
|
|
410
|
+
def _check_rewards_are_named_once(self) -> Self:
|
|
411
|
+
"""Reject a trace that scores the same reward twice."""
|
|
412
|
+
names = [reward.name for reward in self.rewards]
|
|
413
|
+
if len(set(names)) != len(names):
|
|
414
|
+
raise ValueError("a trace records each reward exactly once")
|
|
415
|
+
return self
|
|
416
|
+
|
|
417
|
+
@model_validator(mode="after")
|
|
418
|
+
def _check_sampling_was_resolved(self) -> Self:
|
|
419
|
+
"""Reject a trace that records no sampling settings at all."""
|
|
420
|
+
if not self.sampling:
|
|
421
|
+
raise ValueError(
|
|
422
|
+
"a trace records the sampling settings its rollout resolved"
|
|
423
|
+
)
|
|
424
|
+
return self
|
|
425
|
+
|
|
426
|
+
def reward(self, name: str) -> NormalizedReward | None:
|
|
427
|
+
"""Return one reward by name, or ``None`` when it was not scored."""
|
|
428
|
+
for reward in self.rewards:
|
|
429
|
+
if reward.name == name:
|
|
430
|
+
return reward
|
|
431
|
+
return None
|
|
432
|
+
|
|
433
|
+
|
|
434
|
+
class NormalizedEpisode(ProtocolModel):
|
|
435
|
+
"""One task's episode, ordered by the Campaign's committed membership."""
|
|
436
|
+
|
|
437
|
+
episode_id: NonEmptyString
|
|
438
|
+
env_id: NonEmptyString
|
|
439
|
+
task_hash: Digest
|
|
440
|
+
task_position: int = Field(ge=0)
|
|
441
|
+
ok: bool
|
|
442
|
+
traces: list[NormalizedTrace]
|
|
443
|
+
errors: list[NormalizedExecutionError]
|
|
444
|
+
raw_episode_digest: Digest
|
|
445
|
+
|
|
446
|
+
@model_validator(mode="after")
|
|
447
|
+
def _check_every_trace_belongs_to_this_task(self) -> Self:
|
|
448
|
+
"""Reject an episode whose traces score a different task."""
|
|
449
|
+
for trace in self.traces:
|
|
450
|
+
if trace.task_hash != self.task_hash:
|
|
451
|
+
raise ValueError(
|
|
452
|
+
"an episode's traces all score the episode's own task; got "
|
|
453
|
+
f"{trace.task_hash} inside {self.task_hash}"
|
|
454
|
+
)
|
|
455
|
+
return self
|
|
456
|
+
|
|
457
|
+
|
|
458
|
+
# ---------------------------------------------------------------------------
|
|
459
|
+
# Results
|
|
460
|
+
# ---------------------------------------------------------------------------
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
class VariantExecutionResult(ProtocolModel):
|
|
464
|
+
"""One variant, executed, with raw evidence and its normalized projection."""
|
|
465
|
+
|
|
466
|
+
variant: VariantName
|
|
467
|
+
experiment_manifest_digest: Digest
|
|
468
|
+
resolved_verifiers_config: ArtifactRef
|
|
469
|
+
raw_traces: ArtifactRef
|
|
470
|
+
eval_log: ArtifactRef
|
|
471
|
+
normalized_episodes: ArtifactRef
|
|
472
|
+
child_outcome: ChildProcessOutcome
|
|
473
|
+
image_resolution: SubjectImageResolution
|
|
474
|
+
episodes: list[NormalizedEpisode]
|
|
475
|
+
|
|
476
|
+
@model_validator(mode="after")
|
|
477
|
+
def _check_the_outcome_describes_this_variant(self) -> Self:
|
|
478
|
+
"""Reject a result whose child outcome belongs to the other variant."""
|
|
479
|
+
if self.child_outcome.variant is not self.variant:
|
|
480
|
+
raise ValueError(
|
|
481
|
+
f"a {self.variant.value} result carries a "
|
|
482
|
+
f"{self.child_outcome.variant.value} child outcome"
|
|
483
|
+
)
|
|
484
|
+
if self.image_resolution.variant is not self.variant:
|
|
485
|
+
raise ValueError(
|
|
486
|
+
f"a {self.variant.value} result carries a "
|
|
487
|
+
f"{self.image_resolution.variant.value} image resolution"
|
|
488
|
+
)
|
|
489
|
+
return self
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
class RealExecutionResult(ProtocolModel):
|
|
493
|
+
"""What WP6 hands WP7: both variants, executed under one schedule."""
|
|
494
|
+
|
|
495
|
+
execution_backend: Literal["verifiers"]
|
|
496
|
+
engine_digest: Digest
|
|
497
|
+
verifiers_revision: NonEmptyString
|
|
498
|
+
schedule: VariantSchedule
|
|
499
|
+
baseline: VariantExecutionResult
|
|
500
|
+
candidate: VariantExecutionResult
|
|
501
|
+
|
|
502
|
+
@model_validator(mode="after")
|
|
503
|
+
def _check_each_side_is_the_side_it_claims(self) -> Self:
|
|
504
|
+
"""Reject a result that files a variant under the wrong name."""
|
|
505
|
+
if self.baseline.variant is not VariantName.BASELINE:
|
|
506
|
+
raise ValueError("the baseline slot holds the baseline variant")
|
|
507
|
+
if self.candidate.variant is not VariantName.CANDIDATE:
|
|
508
|
+
raise ValueError("the candidate slot holds the candidate variant")
|
|
509
|
+
return self
|
|
510
|
+
|
|
511
|
+
|
|
512
|
+
# ---------------------------------------------------------------------------
|
|
513
|
+
# Checks
|
|
514
|
+
# ---------------------------------------------------------------------------
|
|
515
|
+
|
|
516
|
+
|
|
517
|
+
class ExecutionCheck(ProtocolModel):
|
|
518
|
+
"""One named question about an execution, and its answer.
|
|
519
|
+
|
|
520
|
+
The same shape as ``ValidationCheck`` (spec section 21.5) and for the same
|
|
521
|
+
reason: a caller reads an ordered list of named verdicts rather than
|
|
522
|
+
catching exceptions to discover which rule failed.
|
|
523
|
+
"""
|
|
524
|
+
|
|
525
|
+
id: NonEmptyString
|
|
526
|
+
status: Literal["passed", "failed", "warning", "not_run"]
|
|
527
|
+
detail: NonEmptyString
|