techtree 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- techtree/__init__.py +35 -0
- techtree/__main__.py +14 -0
- techtree/canonical.py +239 -0
- techtree/catalog/__init__.py +25 -0
- techtree/catalog/repository.py +400 -0
- techtree/catalog/service.py +419 -0
- techtree/cli/__init__.py +1 -0
- techtree/cli/app.py +416 -0
- techtree/cli/commands/__init__.py +1 -0
- techtree/cli/commands/climb.py +1223 -0
- techtree/cli/commands/doctor.py +147 -0
- techtree/cli/commands/engine.py +207 -0
- techtree/cli/commands/proof.py +556 -0
- techtree/cli/commands/publish.py +447 -0
- techtree/cli/commands/release.py +303 -0
- techtree/cli/commands/run.py +1067 -0
- techtree/cli/commands/setup.py +181 -0
- techtree/cli/commands/skill.py +221 -0
- techtree/cli/commands/uplift.py +698 -0
- techtree/cli/commands/withdraw.py +212 -0
- techtree/cli/confirm.py +47 -0
- techtree/cli/context.py +96 -0
- techtree/cli/invoke.py +220 -0
- techtree/cli/output.py +280 -0
- techtree/constants.py +138 -0
- techtree/crypto.py +128 -0
- techtree/doctor/__init__.py +1 -0
- techtree/doctor/checks.py +675 -0
- techtree/doctor/execution_checks.py +435 -0
- techtree/doctor/service.py +326 -0
- techtree/drafts/__init__.py +32 -0
- techtree/drafts/source.py +146 -0
- techtree/drafts/store.py +992 -0
- techtree/engines/__init__.py +1 -0
- techtree/engines/bundle.py +251 -0
- techtree/engines/installer.py +679 -0
- techtree/engines/registry.py +235 -0
- techtree/engines/runner.py +170 -0
- techtree/errors.py +262 -0
- techtree/fs.py +234 -0
- techtree/harness.py +108 -0
- techtree/identity/__init__.py +41 -0
- techtree/identity/models.py +113 -0
- techtree/identity/service.py +199 -0
- techtree/identity/store.py +263 -0
- techtree/ids.py +85 -0
- techtree/manifests/__init__.py +39 -0
- techtree/manifests/builder.py +433 -0
- techtree/manifests/compare.py +376 -0
- techtree/models/__init__.py +282 -0
- techtree/models/base.py +201 -0
- techtree/models/campaign.py +484 -0
- techtree/models/catalog.py +227 -0
- techtree/models/cli.py +151 -0
- techtree/models/climb.py +254 -0
- techtree/models/data_policy.py +130 -0
- techtree/models/engine.py +156 -0
- techtree/models/episode_receipt.py +130 -0
- techtree/models/evaluation_backend.py +113 -0
- techtree/models/experiment.py +154 -0
- techtree/models/run.py +214 -0
- techtree/models/skill.py +156 -0
- techtree/models/uplift_report.py +158 -0
- techtree/models/validation.py +299 -0
- techtree/paths.py +116 -0
- techtree/presentation/__init__.py +31 -0
- techtree/presentation/build.py +1242 -0
- techtree/presentation/compact.py +246 -0
- techtree/presentation/evidence.py +169 -0
- techtree/presentation/models.py +358 -0
- techtree/presentation/rich.py +312 -0
- techtree/presentation/sanitize.py +156 -0
- techtree/publication/__init__.py +44 -0
- techtree/publication/address.py +180 -0
- techtree/publication/coordinates.py +26 -0
- techtree/publication/journal.py +212 -0
- techtree/publication/keccak.py +183 -0
- techtree/publication/models.py +209 -0
- techtree/publication/offer.py +35 -0
- techtree/publication/service.py +618 -0
- techtree/publication/transport.py +296 -0
- techtree/publication/verify.py +242 -0
- techtree/publication/withdraw.py +156 -0
- techtree/py.typed +0 -0
- techtree/receipts/__init__.py +52 -0
- techtree/receipts/bundle.py +578 -0
- techtree/receipts/compare.py +1065 -0
- techtree/receipts/episode.py +672 -0
- techtree/receipts/execution.py +630 -0
- techtree/receipts/observed.py +474 -0
- techtree/receipts/set.py +336 -0
- techtree/receipts/uplift.py +655 -0
- techtree/receipts/verify.py +1055 -0
- techtree/release/__init__.py +9 -0
- techtree/release/bootstrap.py +509 -0
- techtree/release/checks.py +376 -0
- techtree/release/document.py +125 -0
- techtree/release/generate.py +221 -0
- techtree/release/models.py +293 -0
- techtree/release/provenance.py +109 -0
- techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
- techtree/resources/catalog/catalog.json +32 -0
- techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
- techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
- techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
- techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
- techtree/resources/engines/default/engine.json +20 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
- techtree/resources/engines/default/pyproject.toml +23 -0
- techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
- techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
- techtree/resources/engines/default/tools/normalize_validation.py +222 -0
- techtree/resources/engines/default/uv.lock +1758 -0
- techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
- techtree/resources/release/build-provenance.json +4 -0
- techtree/resources/release/release-core.json +24 -0
- techtree/runs/__init__.py +31 -0
- techtree/runs/artifacts.py +750 -0
- techtree/runs/child_registry.py +228 -0
- techtree/runs/events.py +478 -0
- techtree/runs/executor.py +140 -0
- techtree/runs/fake.py +741 -0
- techtree/runs/launcher.py +253 -0
- techtree/runs/machine.py +489 -0
- techtree/runs/real.py +789 -0
- techtree/runs/service.py +616 -0
- techtree/runs/store.py +555 -0
- techtree/runs/validation.py +259 -0
- techtree/runs/variants.py +684 -0
- techtree/settings.py +143 -0
- techtree/skills/__init__.py +14 -0
- techtree/skills/archive.py +282 -0
- techtree/skills/policy.py +62 -0
- techtree/skills/scanner.py +394 -0
- techtree/skills/service.py +752 -0
- techtree/skills/starter.py +434 -0
- techtree/tasksets/__init__.py +1 -0
- techtree/tasksets/membership.py +269 -0
- techtree/tasksets/provider.py +207 -0
- techtree/tasksets/resolver.py +311 -0
- techtree/tasksets/service.py +484 -0
- techtree/tasksets/verifiers_cli.py +538 -0
- techtree/uplift/__init__.py +20 -0
- techtree/uplift/context.py +544 -0
- techtree/uplift/derive.py +203 -0
- techtree/uplift/public_tasks.py +151 -0
- techtree/uplift/service.py +719 -0
- techtree/uplift/source.py +160 -0
- techtree/verifiers/__init__.py +31 -0
- techtree/verifiers/budget.py +219 -0
- techtree/verifiers/child.py +633 -0
- techtree/verifiers/compiler.py +432 -0
- techtree/verifiers/config.py +365 -0
- techtree/verifiers/credentials.py +321 -0
- techtree/verifiers/image.py +126 -0
- techtree/verifiers/models.py +527 -0
- techtree/verifiers/outputs.py +368 -0
- techtree/verifiers/progress.py +192 -0
- techtree/verifiers/supervisor.py +341 -0
- techtree/verifiers/verify.py +782 -0
- techtree/version.py +39 -0
- techtree/worker/__init__.py +18 -0
- techtree/worker/execute.py +487 -0
- techtree/worker/main.py +57 -0
- techtree-0.1.0.dist-info/METADATA +344 -0
- techtree-0.1.0.dist-info/RECORD +174 -0
- techtree-0.1.0.dist-info/WHEEL +4 -0
- techtree-0.1.0.dist-info/entry_points.txt +3 -0
- techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
"""Rights and permitted future uses of Campaign artifacts. Spec section 11.4.
|
|
2
|
+
|
|
3
|
+
A ``DataPolicy`` is required before any episode exists, because rights cannot
|
|
4
|
+
be retrofitted honestly. Once a participant has run a comparison, asking them
|
|
5
|
+
afterwards whether the transcripts may be used for training is asking a
|
|
6
|
+
question whose answer was already assumed.
|
|
7
|
+
|
|
8
|
+
The policy is a plain statement of permissions. It is immutable by digest:
|
|
9
|
+
changing any permission produces a different policy, which produces a
|
|
10
|
+
different ``CampaignSpec``, which is exactly the visibility the rule exists to
|
|
11
|
+
create. Nothing here grants Techtree the ability to do anything: a policy
|
|
12
|
+
that permits something is not a command that does it, and the only thing that
|
|
13
|
+
sends a run anywhere is a person running ``techtree publish``.
|
|
14
|
+
|
|
15
|
+
Contradiction checks between a public Climb and its policy live in
|
|
16
|
+
:mod:`techtree.models.climb`, where both objects are in scope.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from typing import Literal, Self
|
|
22
|
+
|
|
23
|
+
from pydantic import Field, model_validator
|
|
24
|
+
|
|
25
|
+
from techtree.models.base import NonEmptyString, ProtocolModel
|
|
26
|
+
|
|
27
|
+
__all__ = [
|
|
28
|
+
"CandidateSkillPolicy",
|
|
29
|
+
"DataOwner",
|
|
30
|
+
"DataPolicy",
|
|
31
|
+
"DerivedArtifactPolicy",
|
|
32
|
+
"RawEpisodePolicy",
|
|
33
|
+
"RevocationPolicy",
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
type Permission = Literal["allowed", "prohibited", "consent_required"]
|
|
38
|
+
"""Whether a use is permitted outright, forbidden, or gated on fresh consent."""
|
|
39
|
+
|
|
40
|
+
type Visibility = Literal["public", "private", "prohibited"]
|
|
41
|
+
"""Whether a derived artifact may be published, kept, or not produced at all."""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class DataOwner(ProtocolModel):
|
|
45
|
+
"""Who owns the artifacts a Campaign produces."""
|
|
46
|
+
|
|
47
|
+
kind: Literal["participant", "account", "shared"]
|
|
48
|
+
account_ref: NonEmptyString | None = None
|
|
49
|
+
|
|
50
|
+
@model_validator(mode="after")
|
|
51
|
+
def _check_account_reference_matches_ownership(self) -> Self:
|
|
52
|
+
"""Require an account reference exactly where ownership implies one."""
|
|
53
|
+
if self.kind == "account" and self.account_ref is None:
|
|
54
|
+
raise ValueError("account-owned data must name the owning account_ref")
|
|
55
|
+
if self.kind == "participant" and self.account_ref is not None:
|
|
56
|
+
raise ValueError(
|
|
57
|
+
"participant-owned data must not name an account_ref; use the "
|
|
58
|
+
"shared owner kind when an account also holds rights"
|
|
59
|
+
)
|
|
60
|
+
return self
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class RawEpisodePolicy(ProtocolModel):
|
|
64
|
+
"""What may happen to raw episode transcripts."""
|
|
65
|
+
|
|
66
|
+
local_retention: Literal["allowed", "prohibited", "required"]
|
|
67
|
+
server_upload: Permission
|
|
68
|
+
public_release: Permission
|
|
69
|
+
reproduction_access: Permission
|
|
70
|
+
training_use: Permission
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
class DerivedArtifactPolicy(ProtocolModel):
|
|
74
|
+
"""What may happen to everything computed from the episodes."""
|
|
75
|
+
|
|
76
|
+
aggregate_scores: Visibility
|
|
77
|
+
uplift_report: Visibility
|
|
78
|
+
redacted_trace_projection: Visibility
|
|
79
|
+
anonymized_product_analytics: Permission
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class CandidateSkillPolicy(ProtocolModel):
|
|
83
|
+
"""Who owns the submitted skill and whether it can be published."""
|
|
84
|
+
|
|
85
|
+
ownership: Literal["participant", "account", "shared"]
|
|
86
|
+
public_release: Literal[
|
|
87
|
+
"required_for_climb",
|
|
88
|
+
"allowed",
|
|
89
|
+
"prohibited",
|
|
90
|
+
"consent_required",
|
|
91
|
+
]
|
|
92
|
+
training_use: Permission
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
class RevocationPolicy(ProtocolModel):
|
|
96
|
+
"""What a participant can withdraw later, and what stays published."""
|
|
97
|
+
|
|
98
|
+
future_use_revocable: bool
|
|
99
|
+
immutable_published_proofs_remain: bool
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
class DataPolicy(ProtocolModel):
|
|
103
|
+
"""The complete rights statement a Campaign runs under."""
|
|
104
|
+
|
|
105
|
+
schema_version: Literal["techtree.data-policy.v1alpha1"]
|
|
106
|
+
id: NonEmptyString
|
|
107
|
+
version: int = Field(ge=1)
|
|
108
|
+
owner: DataOwner
|
|
109
|
+
raw_episodes: RawEpisodePolicy
|
|
110
|
+
derived_artifacts: DerivedArtifactPolicy
|
|
111
|
+
candidate_skill: CandidateSkillPolicy
|
|
112
|
+
revocation: RevocationPolicy
|
|
113
|
+
|
|
114
|
+
@model_validator(mode="after")
|
|
115
|
+
def _check_internal_consistency(self) -> Self:
|
|
116
|
+
"""Reject a policy that permits a use it also makes impossible."""
|
|
117
|
+
if self.raw_episodes.local_retention == "prohibited":
|
|
118
|
+
for name in (
|
|
119
|
+
"server_upload",
|
|
120
|
+
"public_release",
|
|
121
|
+
"reproduction_access",
|
|
122
|
+
"training_use",
|
|
123
|
+
):
|
|
124
|
+
if getattr(self.raw_episodes, name) != "prohibited":
|
|
125
|
+
raise ValueError(
|
|
126
|
+
f"raw_episodes.{name} cannot be permitted while "
|
|
127
|
+
"local_retention is prohibited; there would be nothing "
|
|
128
|
+
"left to share"
|
|
129
|
+
)
|
|
130
|
+
return self
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
"""The managed Verifiers engine. Spec 11.13, decisions 0003 A8/A9.
|
|
2
|
+
|
|
3
|
+
The engine is a locked, content-addressed bundle: a Python version, a pinned
|
|
4
|
+
Verifiers revision, and the reference packages, described by an
|
|
5
|
+
``EngineDescriptor``. The bundle's digest covers the static files and is not
|
|
6
|
+
stored inside the descriptor, because a document cannot contain its own hash.
|
|
7
|
+
|
|
8
|
+
Host platforms use one vocabulary — Go/OCI-style ``<os>/<arch>`` — everywhere
|
|
9
|
+
they appear. :func:`normalize_host_platform` is the only way a raw
|
|
10
|
+
``sys.platform`` and ``platform.machine()`` pair becomes one of those strings,
|
|
11
|
+
and it refuses rather than guesses on anything else. An unsupported host is a
|
|
12
|
+
prerequisite failure with a name, not an install that fails later for reasons
|
|
13
|
+
the user cannot read.
|
|
14
|
+
|
|
15
|
+
The engine's host platform and the subject runtime's Docker platform share this
|
|
16
|
+
vocabulary but remain separate concepts: one is where Techtree runs, the other
|
|
17
|
+
is where the evaluated agent runs, and they are frequently not the same
|
|
18
|
+
machine.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
from typing import Final, Literal, Self
|
|
24
|
+
|
|
25
|
+
from pydantic import model_validator
|
|
26
|
+
|
|
27
|
+
from techtree.errors import PrerequisiteError
|
|
28
|
+
from techtree.models.base import (
|
|
29
|
+
Digest,
|
|
30
|
+
NonEmptyString,
|
|
31
|
+
ProtocolModel,
|
|
32
|
+
StateModel,
|
|
33
|
+
UtcDateTime,
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
__all__ = [
|
|
37
|
+
"EngineDescriptor",
|
|
38
|
+
"EngineInstallation",
|
|
39
|
+
"EnginePackage",
|
|
40
|
+
"EngineStatus",
|
|
41
|
+
"HostPlatform",
|
|
42
|
+
"normalize_host_platform",
|
|
43
|
+
]
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
type HostPlatform = Literal[
|
|
47
|
+
"darwin/amd64",
|
|
48
|
+
"darwin/arm64",
|
|
49
|
+
"linux/amd64",
|
|
50
|
+
"linux/arm64",
|
|
51
|
+
]
|
|
52
|
+
"""The complete host-platform vocabulary. Decisions document 0003 A9."""
|
|
53
|
+
|
|
54
|
+
#: Every machine string that means the same architecture. The keys are what
|
|
55
|
+
#: ``platform.machine()`` reports across the platforms Techtree supports.
|
|
56
|
+
_ARCHITECTURES: Final[dict[str, str]] = {
|
|
57
|
+
"aarch64": "arm64",
|
|
58
|
+
"amd64": "amd64",
|
|
59
|
+
"arm64": "arm64",
|
|
60
|
+
"x86_64": "amd64",
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
#: The closed vocabulary, keyed by the pair it is derived from. A table rather
|
|
64
|
+
#: than string concatenation, so that the only values this function can return
|
|
65
|
+
#: are the four the protocol defines.
|
|
66
|
+
_HOST_PLATFORMS: Final[dict[tuple[str, str], HostPlatform]] = {
|
|
67
|
+
("darwin", "amd64"): "darwin/amd64",
|
|
68
|
+
("darwin", "arm64"): "darwin/arm64",
|
|
69
|
+
("linux", "amd64"): "linux/amd64",
|
|
70
|
+
("linux", "arm64"): "linux/arm64",
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def normalize_host_platform(sys_platform: str, machine: str) -> HostPlatform:
|
|
75
|
+
"""Return the ``<os>/<arch>`` name for a host, or refuse.
|
|
76
|
+
|
|
77
|
+
Raises :class:`~techtree.errors.PrerequisiteError` for any combination
|
|
78
|
+
outside the supported vocabulary. Guessing would produce an engine install
|
|
79
|
+
that resolves wheels for the wrong architecture and fails much later,
|
|
80
|
+
somewhere far less legible than here.
|
|
81
|
+
"""
|
|
82
|
+
operating_system = sys_platform.strip().lower()
|
|
83
|
+
architecture = _ARCHITECTURES.get(machine.strip().lower(), "")
|
|
84
|
+
normalized = _HOST_PLATFORMS.get((operating_system, architecture))
|
|
85
|
+
|
|
86
|
+
if normalized is None:
|
|
87
|
+
raise PrerequisiteError(
|
|
88
|
+
f"unsupported host platform {sys_platform}/{machine}; Techtree "
|
|
89
|
+
"supports darwin and linux on arm64 and amd64",
|
|
90
|
+
code="unsupported_host_platform",
|
|
91
|
+
details={"sys_platform": sys_platform, "machine": machine},
|
|
92
|
+
)
|
|
93
|
+
return normalized
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class EnginePackage(ProtocolModel):
|
|
97
|
+
"""One package shipped inside the engine bundle."""
|
|
98
|
+
|
|
99
|
+
name: NonEmptyString
|
|
100
|
+
version: NonEmptyString
|
|
101
|
+
source_digest: Digest
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
class EngineDescriptor(ProtocolModel):
|
|
105
|
+
"""What one engine bundle is, without saying what it hashes to."""
|
|
106
|
+
|
|
107
|
+
schema_version: Literal["techtree.engine.v1alpha1"]
|
|
108
|
+
name: NonEmptyString
|
|
109
|
+
python_version: NonEmptyString
|
|
110
|
+
verifiers_version: NonEmptyString
|
|
111
|
+
verifiers_revision: NonEmptyString
|
|
112
|
+
supported_hosts: list[HostPlatform]
|
|
113
|
+
packages: list[EnginePackage]
|
|
114
|
+
|
|
115
|
+
@model_validator(mode="after")
|
|
116
|
+
def _check_hosts_and_packages_are_listed_once(self) -> Self:
|
|
117
|
+
"""Reject an empty or repeating host or package list."""
|
|
118
|
+
if not self.supported_hosts:
|
|
119
|
+
raise ValueError("an engine must support at least one host platform")
|
|
120
|
+
if len(set(self.supported_hosts)) != len(self.supported_hosts):
|
|
121
|
+
raise ValueError("supported_hosts must not repeat a platform")
|
|
122
|
+
names = [package.name for package in self.packages]
|
|
123
|
+
if len(set(names)) != len(names):
|
|
124
|
+
raise ValueError("an engine ships each package exactly once")
|
|
125
|
+
return self
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
class EngineInstallation(StateModel):
|
|
129
|
+
"""A locally installed engine, as the registry records it."""
|
|
130
|
+
|
|
131
|
+
digest: Digest
|
|
132
|
+
installed_at: UtcDateTime
|
|
133
|
+
python_executable: NonEmptyString
|
|
134
|
+
descriptor_digest: Digest
|
|
135
|
+
verified: bool
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
class EngineStatus(ProtocolModel):
|
|
139
|
+
"""What the CLI reports about one engine."""
|
|
140
|
+
|
|
141
|
+
digest: Digest
|
|
142
|
+
installed: bool
|
|
143
|
+
active: bool
|
|
144
|
+
verified: bool
|
|
145
|
+
path: NonEmptyString
|
|
146
|
+
python_executable: NonEmptyString | None
|
|
147
|
+
detail: NonEmptyString
|
|
148
|
+
|
|
149
|
+
@model_validator(mode="after")
|
|
150
|
+
def _check_status_is_coherent(self) -> Self:
|
|
151
|
+
"""Reject a status that verifies or activates an engine that is absent."""
|
|
152
|
+
if not self.installed and (self.verified or self.active):
|
|
153
|
+
raise ValueError(
|
|
154
|
+
"an engine that is not installed cannot be active or verified"
|
|
155
|
+
)
|
|
156
|
+
return self
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
"""What one episode produced. Spec section 11.10.
|
|
2
|
+
|
|
3
|
+
The shape is frozen now and fake-populated until WP6, which is the point: the
|
|
4
|
+
statuses that say "this number is not evidence" have to exist before the first
|
|
5
|
+
number does. ``development_only`` is a first-class score and evidence status
|
|
6
|
+
rather than an absent field, so a fake receipt is unmistakably fake to a reader
|
|
7
|
+
and to a service, and nothing has to infer it from context.
|
|
8
|
+
|
|
9
|
+
A receipt points at ``campaign_spec_digest`` and carries the data policy that
|
|
10
|
+
governed it. The public Climb is an optional context, never the anchor: a
|
|
11
|
+
reproduction run has no Climb, and its receipts must still be complete.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from enum import StrEnum
|
|
17
|
+
from typing import Literal, Self
|
|
18
|
+
|
|
19
|
+
from pydantic import model_validator
|
|
20
|
+
|
|
21
|
+
from techtree.models.base import (
|
|
22
|
+
ArtifactRef,
|
|
23
|
+
Digest,
|
|
24
|
+
NonEmptyString,
|
|
25
|
+
ProtocolModel,
|
|
26
|
+
)
|
|
27
|
+
from techtree.models.campaign import ProgramRef, PublicContext
|
|
28
|
+
from techtree.models.evaluation_backend import EvaluationBackendSpec
|
|
29
|
+
from techtree.models.experiment import ExperimentVariant
|
|
30
|
+
|
|
31
|
+
__all__ = [
|
|
32
|
+
"EpisodeReceipt",
|
|
33
|
+
"EvidenceStatus",
|
|
34
|
+
"NamedTraceReceipt",
|
|
35
|
+
"ScoreStatus",
|
|
36
|
+
"SubjectRuntimeReceipt",
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class ScoreStatus(StrEnum):
|
|
41
|
+
"""How much weight the recorded reward carries."""
|
|
42
|
+
|
|
43
|
+
PENDING = "pending"
|
|
44
|
+
VALID = "valid"
|
|
45
|
+
INVALID = "invalid"
|
|
46
|
+
ERRORED = "errored"
|
|
47
|
+
MISSING = "missing"
|
|
48
|
+
DEVELOPMENT_ONLY = "development_only"
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class EvidenceStatus(StrEnum):
|
|
52
|
+
"""How complete the supporting evidence is."""
|
|
53
|
+
|
|
54
|
+
NOT_COLLECTED = "not_collected"
|
|
55
|
+
COMPLETE = "complete"
|
|
56
|
+
PARTIAL = "partial"
|
|
57
|
+
INVALID = "invalid"
|
|
58
|
+
DEVELOPMENT_ONLY = "development_only"
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class NamedTraceReceipt(ProtocolModel):
|
|
62
|
+
"""One named trace within an episode."""
|
|
63
|
+
|
|
64
|
+
role: NonEmptyString
|
|
65
|
+
trace_id: NonEmptyString
|
|
66
|
+
trace_digest: Digest
|
|
67
|
+
task_hash: Digest
|
|
68
|
+
rewards: dict[str, float]
|
|
69
|
+
metrics: dict[str, float | None]
|
|
70
|
+
ok: bool
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
class SubjectRuntimeReceipt(ProtocolModel):
|
|
74
|
+
"""Where the subject agent actually executed, if it executed."""
|
|
75
|
+
|
|
76
|
+
kind: Literal["not_executed", "docker"]
|
|
77
|
+
resolved_image_digest: Digest | None = None
|
|
78
|
+
platform: NonEmptyString | None = None
|
|
79
|
+
|
|
80
|
+
@model_validator(mode="after")
|
|
81
|
+
def _check_runtime_evidence_matches_kind(self) -> Self:
|
|
82
|
+
"""Reject runtime detail on an episode that never ran a runtime."""
|
|
83
|
+
if self.kind == "not_executed" and (
|
|
84
|
+
self.resolved_image_digest is not None or self.platform is not None
|
|
85
|
+
):
|
|
86
|
+
raise ValueError(
|
|
87
|
+
"an episode that did not execute a runtime cannot report the "
|
|
88
|
+
"image or platform it executed on"
|
|
89
|
+
)
|
|
90
|
+
if self.kind == "docker" and self.resolved_image_digest is None:
|
|
91
|
+
raise ValueError("a docker episode records the image digest it ran")
|
|
92
|
+
return self
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
class EpisodeReceipt(ProtocolModel):
|
|
96
|
+
"""The complete record of one scored episode."""
|
|
97
|
+
|
|
98
|
+
schema_version: Literal["techtree.episode-receipt.v1alpha1"]
|
|
99
|
+
id: NonEmptyString
|
|
100
|
+
run_id: NonEmptyString
|
|
101
|
+
campaign_spec_digest: Digest
|
|
102
|
+
program_ref: ProgramRef | None
|
|
103
|
+
public_context: PublicContext | None
|
|
104
|
+
data_policy_digest: Digest
|
|
105
|
+
outcome_contract_digest: Digest | None
|
|
106
|
+
evaluation_backend: EvaluationBackendSpec
|
|
107
|
+
subject_runtime: SubjectRuntimeReceipt
|
|
108
|
+
variant: ExperimentVariant
|
|
109
|
+
experiment_manifest_digest: Digest
|
|
110
|
+
episode_id: NonEmptyString
|
|
111
|
+
episode_digest: Digest
|
|
112
|
+
task_hash: Digest
|
|
113
|
+
named_traces: dict[str, list[NamedTraceReceipt]]
|
|
114
|
+
score_status: ScoreStatus
|
|
115
|
+
evidence_status: EvidenceStatus
|
|
116
|
+
execution_backend: Literal["fake", "verifiers"]
|
|
117
|
+
artifacts: list[ArtifactRef]
|
|
118
|
+
|
|
119
|
+
@model_validator(mode="after")
|
|
120
|
+
def _check_fake_episodes_are_unmistakably_fake(self) -> Self:
|
|
121
|
+
"""Refuse to let a fake episode wear a real score."""
|
|
122
|
+
if self.execution_backend == "fake" and (
|
|
123
|
+
self.score_status is not ScoreStatus.DEVELOPMENT_ONLY
|
|
124
|
+
or self.evidence_status is not EvidenceStatus.DEVELOPMENT_ONLY
|
|
125
|
+
):
|
|
126
|
+
raise ValueError(
|
|
127
|
+
"a fake episode reports development_only score and evidence; "
|
|
128
|
+
"any other status would present invented numbers as results"
|
|
129
|
+
)
|
|
130
|
+
return self
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
"""Who orchestrated and attested to an evaluation. Spec section 11.3.
|
|
2
|
+
|
|
3
|
+
The evaluation backend and the subject runtime answer two different questions
|
|
4
|
+
and are deliberately separate objects:
|
|
5
|
+
|
|
6
|
+
``EvaluationBackendSpec``
|
|
7
|
+
Who ran the comparison and whose word the result rests on.
|
|
8
|
+
|
|
9
|
+
``RuntimeSpec`` (in :mod:`techtree.models.campaign`)
|
|
10
|
+
Where the evaluated agent's process actually executed.
|
|
11
|
+
|
|
12
|
+
Conflating them is how a self-reported local result quietly acquires the
|
|
13
|
+
authority of a platform-attested one. Each backend kind therefore fixes the
|
|
14
|
+
attestation it is allowed to claim, and the references that must or must not
|
|
15
|
+
accompany it.
|
|
16
|
+
|
|
17
|
+
The enum keeps the future backends so that stored documents and published
|
|
18
|
+
schemas do not need a version bump when they arrive. Services enforce the
|
|
19
|
+
narrower WP0–WP5 rule — only ``local_techtree`` — at the point of use, not
|
|
20
|
+
here, because a document that merely mentions a future backend must still be
|
|
21
|
+
parseable.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
from enum import StrEnum
|
|
27
|
+
from typing import Literal, Self
|
|
28
|
+
|
|
29
|
+
from pydantic import model_validator
|
|
30
|
+
|
|
31
|
+
from techtree.models.base import NonEmptyString, ProtocolModel
|
|
32
|
+
|
|
33
|
+
__all__ = [
|
|
34
|
+
"SUPPORTED_EVALUATION_BACKEND_KINDS",
|
|
35
|
+
"AttestationKind",
|
|
36
|
+
"EvaluationBackendKind",
|
|
37
|
+
"EvaluationBackendSpec",
|
|
38
|
+
]
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class EvaluationBackendKind(StrEnum):
|
|
42
|
+
"""Which system orchestrated the evaluation."""
|
|
43
|
+
|
|
44
|
+
LOCAL_TECHTREE = "local_techtree"
|
|
45
|
+
PRIME_LAB = "prime_lab"
|
|
46
|
+
INDEPENDENT_REPRODUCER = "independent_reproducer"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class AttestationKind(StrEnum):
|
|
50
|
+
"""Whose attestation the recorded result carries."""
|
|
51
|
+
|
|
52
|
+
PARTICIPANT = "participant"
|
|
53
|
+
PLATFORM = "platform"
|
|
54
|
+
INDEPENDENT = "independent"
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
#: The only backend kinds WP0–WP5 services accept. The schema is wider than
|
|
58
|
+
#: this on purpose; the runtime surface is not.
|
|
59
|
+
SUPPORTED_EVALUATION_BACKEND_KINDS: frozenset[EvaluationBackendKind] = frozenset(
|
|
60
|
+
{EvaluationBackendKind.LOCAL_TECHTREE}
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class EvaluationBackendSpec(ProtocolModel):
|
|
65
|
+
"""The orchestrating backend and the attestation it carries."""
|
|
66
|
+
|
|
67
|
+
schema_version: Literal["techtree.evaluation-backend.v1alpha1"]
|
|
68
|
+
kind: EvaluationBackendKind
|
|
69
|
+
attestation: AttestationKind
|
|
70
|
+
workspace_ref: NonEmptyString | None = None
|
|
71
|
+
provider_run_ref: NonEmptyString | None = None
|
|
72
|
+
executor_identity: NonEmptyString | None = None
|
|
73
|
+
|
|
74
|
+
@model_validator(mode="after")
|
|
75
|
+
def _check_kind_agrees_with_its_evidence(self) -> Self:
|
|
76
|
+
"""Reject attestation and reference combinations a kind cannot support."""
|
|
77
|
+
if self.kind is EvaluationBackendKind.LOCAL_TECHTREE:
|
|
78
|
+
if self.attestation is not AttestationKind.PARTICIPANT:
|
|
79
|
+
raise ValueError(
|
|
80
|
+
"local_techtree evaluation is self-reported, so its "
|
|
81
|
+
"attestation must be participant"
|
|
82
|
+
)
|
|
83
|
+
if self.workspace_ref is not None:
|
|
84
|
+
raise ValueError("local_techtree evaluation has no workspace_ref")
|
|
85
|
+
if self.provider_run_ref is not None:
|
|
86
|
+
raise ValueError("local_techtree evaluation has no provider_run_ref")
|
|
87
|
+
# executor_identity stays optional: WP0-WP5 record no identity.
|
|
88
|
+
return self
|
|
89
|
+
|
|
90
|
+
if self.kind is EvaluationBackendKind.PRIME_LAB:
|
|
91
|
+
if self.attestation is not AttestationKind.PLATFORM:
|
|
92
|
+
raise ValueError(
|
|
93
|
+
"prime_lab evaluation is platform-attested, so its "
|
|
94
|
+
"attestation must be platform"
|
|
95
|
+
)
|
|
96
|
+
if self.workspace_ref is None and self.provider_run_ref is None:
|
|
97
|
+
raise ValueError(
|
|
98
|
+
"prime_lab evaluation must carry a workspace_ref or a "
|
|
99
|
+
"provider_run_ref so the platform record can be found"
|
|
100
|
+
)
|
|
101
|
+
return self
|
|
102
|
+
|
|
103
|
+
if self.attestation is not AttestationKind.INDEPENDENT:
|
|
104
|
+
raise ValueError(
|
|
105
|
+
"independent_reproducer evaluation must carry an independent "
|
|
106
|
+
"attestation"
|
|
107
|
+
)
|
|
108
|
+
if self.executor_identity is None:
|
|
109
|
+
raise ValueError(
|
|
110
|
+
"independent_reproducer evaluation must name the executor_identity "
|
|
111
|
+
"that stands behind it"
|
|
112
|
+
)
|
|
113
|
+
return self
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
"""One resolved variant of a Campaign, and the comparison. Spec section 11.8.
|
|
2
|
+
|
|
3
|
+
An ``ExperimentManifest`` is a Campaign with every choice made: this taskset,
|
|
4
|
+
this agent, this skill list. Two of them — baseline and candidate — are what a
|
|
5
|
+
run actually executes.
|
|
6
|
+
|
|
7
|
+
The comparison deliberately looks at ``configuration`` and nothing else. A
|
|
8
|
+
manifest identifier, a variant name, a creation time, and a public context all
|
|
9
|
+
differ between two correct manifests of the same experiment; comparing them
|
|
10
|
+
would report differences that mean nothing and bury the one difference that
|
|
11
|
+
means everything. ``ManifestComparison`` therefore compares the configurations
|
|
12
|
+
and reports whether the only difference found is the one the mutation contract
|
|
13
|
+
permits.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from enum import StrEnum
|
|
19
|
+
from typing import Literal, Self
|
|
20
|
+
|
|
21
|
+
from pydantic import model_validator
|
|
22
|
+
|
|
23
|
+
from techtree.models.base import (
|
|
24
|
+
Digest,
|
|
25
|
+
JsonValue,
|
|
26
|
+
NonEmptyString,
|
|
27
|
+
ProtocolModel,
|
|
28
|
+
UtcDateTime,
|
|
29
|
+
)
|
|
30
|
+
from techtree.models.campaign import (
|
|
31
|
+
AgentSpec,
|
|
32
|
+
BudgetSpec,
|
|
33
|
+
CampaignTaskset,
|
|
34
|
+
EnvironmentSpec,
|
|
35
|
+
EvidenceRequirements,
|
|
36
|
+
ExecutionSpec,
|
|
37
|
+
MutationContract,
|
|
38
|
+
MutationKind,
|
|
39
|
+
ProgramRef,
|
|
40
|
+
PublicContext,
|
|
41
|
+
ScoringSpec,
|
|
42
|
+
)
|
|
43
|
+
from techtree.models.evaluation_backend import EvaluationBackendSpec
|
|
44
|
+
|
|
45
|
+
__all__ = [
|
|
46
|
+
"ExperimentConfiguration",
|
|
47
|
+
"ExperimentManifest",
|
|
48
|
+
"ExperimentVariant",
|
|
49
|
+
"JsonDifference",
|
|
50
|
+
"ManifestComparison",
|
|
51
|
+
]
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class ExperimentVariant(StrEnum):
|
|
55
|
+
"""Which side of the comparison a manifest describes."""
|
|
56
|
+
|
|
57
|
+
BASELINE = "baseline"
|
|
58
|
+
CANDIDATE = "candidate"
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class ExperimentConfiguration(ProtocolModel):
|
|
62
|
+
"""The part of a manifest that is compared.
|
|
63
|
+
|
|
64
|
+
This mirrors the Campaign's scientific fields with the choices resolved. It
|
|
65
|
+
holds no identifier, no timestamp, and no public context, precisely so that
|
|
66
|
+
two manifests of the same experiment have byte-identical configurations
|
|
67
|
+
except where the mutation contract allows them to differ.
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
taskset: CampaignTaskset
|
|
71
|
+
environment: EnvironmentSpec
|
|
72
|
+
agents: dict[str, AgentSpec]
|
|
73
|
+
mutation_contract: MutationContract
|
|
74
|
+
evaluation_backend: EvaluationBackendSpec
|
|
75
|
+
execution: ExecutionSpec
|
|
76
|
+
scoring: ScoringSpec
|
|
77
|
+
evidence: EvidenceRequirements
|
|
78
|
+
budgets: BudgetSpec
|
|
79
|
+
data_policy_digest: Digest
|
|
80
|
+
outcome_contract_digest: Digest | None
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
class ExperimentManifest(ProtocolModel):
|
|
84
|
+
"""One fully resolved variant derived from a Campaign."""
|
|
85
|
+
|
|
86
|
+
schema_version: Literal["techtree.experiment.v1alpha1"]
|
|
87
|
+
id: NonEmptyString
|
|
88
|
+
campaign_spec_digest: Digest
|
|
89
|
+
program_ref: ProgramRef | None
|
|
90
|
+
public_context: PublicContext | None
|
|
91
|
+
variant: ExperimentVariant
|
|
92
|
+
configuration: ExperimentConfiguration
|
|
93
|
+
configuration_digest: Digest
|
|
94
|
+
created_at: UtcDateTime
|
|
95
|
+
|
|
96
|
+
@model_validator(mode="after")
|
|
97
|
+
def _check_variant_skill_count(self) -> Self:
|
|
98
|
+
"""Hold each variant to the skill count its mutation kind requires."""
|
|
99
|
+
subject = self.configuration.agents.get("subject")
|
|
100
|
+
if subject is None:
|
|
101
|
+
raise ValueError("an experiment configuration defines a subject agent")
|
|
102
|
+
skills = len(subject.harness.skills)
|
|
103
|
+
|
|
104
|
+
if self.variant is ExperimentVariant.CANDIDATE:
|
|
105
|
+
if skills != 1:
|
|
106
|
+
raise ValueError("the candidate variant carries exactly one skill")
|
|
107
|
+
return self
|
|
108
|
+
|
|
109
|
+
# What the baseline carries is what the candidate is measured against,
|
|
110
|
+
# and the mutation kind is what says which that is (spec section 3.1):
|
|
111
|
+
# nothing for an insertion, the skill being revised for a replacement.
|
|
112
|
+
# The two sides' root digests are a property of the pair rather than of
|
|
113
|
+
# one manifest, so they are checked where the pair is — the builder when
|
|
114
|
+
# one is derived, ``manifests.compare`` when two are read back.
|
|
115
|
+
if self.configuration.mutation_contract.kind is MutationKind.SKILL_INSERTION:
|
|
116
|
+
if skills != 0:
|
|
117
|
+
raise ValueError("the baseline variant carries no candidate skill")
|
|
118
|
+
elif skills != 1:
|
|
119
|
+
raise ValueError(
|
|
120
|
+
"the baseline variant of a skill_replacement carries exactly one "
|
|
121
|
+
"skill to replace"
|
|
122
|
+
)
|
|
123
|
+
return self
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
class JsonDifference(ProtocolModel):
|
|
127
|
+
"""One JSON Pointer at which two configurations disagree."""
|
|
128
|
+
|
|
129
|
+
pointer: NonEmptyString
|
|
130
|
+
baseline: JsonValue | None
|
|
131
|
+
candidate: JsonValue | None
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
class ManifestComparison(ProtocolModel):
|
|
135
|
+
"""Whether the candidate differs from the baseline only where permitted."""
|
|
136
|
+
|
|
137
|
+
baseline_configuration_digest: Digest
|
|
138
|
+
candidate_configuration_digest: Digest
|
|
139
|
+
differences: list[JsonDifference]
|
|
140
|
+
allowed_differences: list[NonEmptyString]
|
|
141
|
+
controlled: bool
|
|
142
|
+
violations: list[NonEmptyString]
|
|
143
|
+
|
|
144
|
+
@model_validator(mode="after")
|
|
145
|
+
def _check_control_agrees_with_violations(self) -> Self:
|
|
146
|
+
"""Reject a comparison that calls itself controlled while listing faults."""
|
|
147
|
+
if self.controlled and self.violations:
|
|
148
|
+
raise ValueError(
|
|
149
|
+
"a controlled comparison has no violations; listing both leaves "
|
|
150
|
+
"a reader unable to tell which one is true"
|
|
151
|
+
)
|
|
152
|
+
if not self.controlled and not self.violations:
|
|
153
|
+
raise ValueError("an uncontrolled comparison must say which rule it broke")
|
|
154
|
+
return self
|