techtree 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. techtree/__init__.py +35 -0
  2. techtree/__main__.py +14 -0
  3. techtree/canonical.py +239 -0
  4. techtree/catalog/__init__.py +25 -0
  5. techtree/catalog/repository.py +400 -0
  6. techtree/catalog/service.py +419 -0
  7. techtree/cli/__init__.py +1 -0
  8. techtree/cli/app.py +416 -0
  9. techtree/cli/commands/__init__.py +1 -0
  10. techtree/cli/commands/climb.py +1223 -0
  11. techtree/cli/commands/doctor.py +147 -0
  12. techtree/cli/commands/engine.py +207 -0
  13. techtree/cli/commands/proof.py +556 -0
  14. techtree/cli/commands/publish.py +447 -0
  15. techtree/cli/commands/release.py +303 -0
  16. techtree/cli/commands/run.py +1067 -0
  17. techtree/cli/commands/setup.py +181 -0
  18. techtree/cli/commands/skill.py +221 -0
  19. techtree/cli/commands/uplift.py +698 -0
  20. techtree/cli/commands/withdraw.py +212 -0
  21. techtree/cli/confirm.py +47 -0
  22. techtree/cli/context.py +96 -0
  23. techtree/cli/invoke.py +220 -0
  24. techtree/cli/output.py +280 -0
  25. techtree/constants.py +138 -0
  26. techtree/crypto.py +128 -0
  27. techtree/doctor/__init__.py +1 -0
  28. techtree/doctor/checks.py +675 -0
  29. techtree/doctor/execution_checks.py +435 -0
  30. techtree/doctor/service.py +326 -0
  31. techtree/drafts/__init__.py +32 -0
  32. techtree/drafts/source.py +146 -0
  33. techtree/drafts/store.py +992 -0
  34. techtree/engines/__init__.py +1 -0
  35. techtree/engines/bundle.py +251 -0
  36. techtree/engines/installer.py +679 -0
  37. techtree/engines/registry.py +235 -0
  38. techtree/engines/runner.py +170 -0
  39. techtree/errors.py +262 -0
  40. techtree/fs.py +234 -0
  41. techtree/harness.py +108 -0
  42. techtree/identity/__init__.py +41 -0
  43. techtree/identity/models.py +113 -0
  44. techtree/identity/service.py +199 -0
  45. techtree/identity/store.py +263 -0
  46. techtree/ids.py +85 -0
  47. techtree/manifests/__init__.py +39 -0
  48. techtree/manifests/builder.py +433 -0
  49. techtree/manifests/compare.py +376 -0
  50. techtree/models/__init__.py +282 -0
  51. techtree/models/base.py +201 -0
  52. techtree/models/campaign.py +484 -0
  53. techtree/models/catalog.py +227 -0
  54. techtree/models/cli.py +151 -0
  55. techtree/models/climb.py +254 -0
  56. techtree/models/data_policy.py +130 -0
  57. techtree/models/engine.py +156 -0
  58. techtree/models/episode_receipt.py +130 -0
  59. techtree/models/evaluation_backend.py +113 -0
  60. techtree/models/experiment.py +154 -0
  61. techtree/models/run.py +214 -0
  62. techtree/models/skill.py +156 -0
  63. techtree/models/uplift_report.py +158 -0
  64. techtree/models/validation.py +299 -0
  65. techtree/paths.py +116 -0
  66. techtree/presentation/__init__.py +31 -0
  67. techtree/presentation/build.py +1242 -0
  68. techtree/presentation/compact.py +246 -0
  69. techtree/presentation/evidence.py +169 -0
  70. techtree/presentation/models.py +358 -0
  71. techtree/presentation/rich.py +312 -0
  72. techtree/presentation/sanitize.py +156 -0
  73. techtree/publication/__init__.py +44 -0
  74. techtree/publication/address.py +180 -0
  75. techtree/publication/coordinates.py +26 -0
  76. techtree/publication/journal.py +212 -0
  77. techtree/publication/keccak.py +183 -0
  78. techtree/publication/models.py +209 -0
  79. techtree/publication/offer.py +35 -0
  80. techtree/publication/service.py +618 -0
  81. techtree/publication/transport.py +296 -0
  82. techtree/publication/verify.py +242 -0
  83. techtree/publication/withdraw.py +156 -0
  84. techtree/py.typed +0 -0
  85. techtree/receipts/__init__.py +52 -0
  86. techtree/receipts/bundle.py +578 -0
  87. techtree/receipts/compare.py +1065 -0
  88. techtree/receipts/episode.py +672 -0
  89. techtree/receipts/execution.py +630 -0
  90. techtree/receipts/observed.py +474 -0
  91. techtree/receipts/set.py +336 -0
  92. techtree/receipts/uplift.py +655 -0
  93. techtree/receipts/verify.py +1055 -0
  94. techtree/release/__init__.py +9 -0
  95. techtree/release/bootstrap.py +509 -0
  96. techtree/release/checks.py +376 -0
  97. techtree/release/document.py +125 -0
  98. techtree/release/generate.py +221 -0
  99. techtree/release/models.py +293 -0
  100. techtree/release/provenance.py +109 -0
  101. techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
  102. techtree/resources/catalog/catalog.json +32 -0
  103. techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
  104. techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
  105. techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
  106. techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
  107. techtree/resources/engines/default/engine.json +20 -0
  108. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
  109. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
  110. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
  111. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
  112. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
  113. techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
  114. techtree/resources/engines/default/pyproject.toml +23 -0
  115. techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
  116. techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
  117. techtree/resources/engines/default/tools/normalize_validation.py +222 -0
  118. techtree/resources/engines/default/uv.lock +1758 -0
  119. techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
  120. techtree/resources/release/build-provenance.json +4 -0
  121. techtree/resources/release/release-core.json +24 -0
  122. techtree/runs/__init__.py +31 -0
  123. techtree/runs/artifacts.py +750 -0
  124. techtree/runs/child_registry.py +228 -0
  125. techtree/runs/events.py +478 -0
  126. techtree/runs/executor.py +140 -0
  127. techtree/runs/fake.py +741 -0
  128. techtree/runs/launcher.py +253 -0
  129. techtree/runs/machine.py +489 -0
  130. techtree/runs/real.py +789 -0
  131. techtree/runs/service.py +616 -0
  132. techtree/runs/store.py +555 -0
  133. techtree/runs/validation.py +259 -0
  134. techtree/runs/variants.py +684 -0
  135. techtree/settings.py +143 -0
  136. techtree/skills/__init__.py +14 -0
  137. techtree/skills/archive.py +282 -0
  138. techtree/skills/policy.py +62 -0
  139. techtree/skills/scanner.py +394 -0
  140. techtree/skills/service.py +752 -0
  141. techtree/skills/starter.py +434 -0
  142. techtree/tasksets/__init__.py +1 -0
  143. techtree/tasksets/membership.py +269 -0
  144. techtree/tasksets/provider.py +207 -0
  145. techtree/tasksets/resolver.py +311 -0
  146. techtree/tasksets/service.py +484 -0
  147. techtree/tasksets/verifiers_cli.py +538 -0
  148. techtree/uplift/__init__.py +20 -0
  149. techtree/uplift/context.py +544 -0
  150. techtree/uplift/derive.py +203 -0
  151. techtree/uplift/public_tasks.py +151 -0
  152. techtree/uplift/service.py +719 -0
  153. techtree/uplift/source.py +160 -0
  154. techtree/verifiers/__init__.py +31 -0
  155. techtree/verifiers/budget.py +219 -0
  156. techtree/verifiers/child.py +633 -0
  157. techtree/verifiers/compiler.py +432 -0
  158. techtree/verifiers/config.py +365 -0
  159. techtree/verifiers/credentials.py +321 -0
  160. techtree/verifiers/image.py +126 -0
  161. techtree/verifiers/models.py +527 -0
  162. techtree/verifiers/outputs.py +368 -0
  163. techtree/verifiers/progress.py +192 -0
  164. techtree/verifiers/supervisor.py +341 -0
  165. techtree/verifiers/verify.py +782 -0
  166. techtree/version.py +39 -0
  167. techtree/worker/__init__.py +18 -0
  168. techtree/worker/execute.py +487 -0
  169. techtree/worker/main.py +57 -0
  170. techtree-0.1.0.dist-info/METADATA +344 -0
  171. techtree-0.1.0.dist-info/RECORD +174 -0
  172. techtree-0.1.0.dist-info/WHEEL +4 -0
  173. techtree-0.1.0.dist-info/entry_points.txt +3 -0
  174. techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,130 @@
1
+ """Rights and permitted future uses of Campaign artifacts. Spec section 11.4.
2
+
3
+ A ``DataPolicy`` is required before any episode exists, because rights cannot
4
+ be retrofitted honestly. Once a participant has run a comparison, asking them
5
+ afterwards whether the transcripts may be used for training is asking a
6
+ question whose answer was already assumed.
7
+
8
+ The policy is a plain statement of permissions. It is immutable by digest:
9
+ changing any permission produces a different policy, which produces a
10
+ different ``CampaignSpec``, which is exactly the visibility the rule exists to
11
+ create. Nothing here grants Techtree the ability to do anything: a policy
12
+ that permits something is not a command that does it, and the only thing that
13
+ sends a run anywhere is a person running ``techtree publish``.
14
+
15
+ Contradiction checks between a public Climb and its policy live in
16
+ :mod:`techtree.models.climb`, where both objects are in scope.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ from typing import Literal, Self
22
+
23
+ from pydantic import Field, model_validator
24
+
25
+ from techtree.models.base import NonEmptyString, ProtocolModel
26
+
27
+ __all__ = [
28
+ "CandidateSkillPolicy",
29
+ "DataOwner",
30
+ "DataPolicy",
31
+ "DerivedArtifactPolicy",
32
+ "RawEpisodePolicy",
33
+ "RevocationPolicy",
34
+ ]
35
+
36
+
37
+ type Permission = Literal["allowed", "prohibited", "consent_required"]
38
+ """Whether a use is permitted outright, forbidden, or gated on fresh consent."""
39
+
40
+ type Visibility = Literal["public", "private", "prohibited"]
41
+ """Whether a derived artifact may be published, kept, or not produced at all."""
42
+
43
+
44
+ class DataOwner(ProtocolModel):
45
+ """Who owns the artifacts a Campaign produces."""
46
+
47
+ kind: Literal["participant", "account", "shared"]
48
+ account_ref: NonEmptyString | None = None
49
+
50
+ @model_validator(mode="after")
51
+ def _check_account_reference_matches_ownership(self) -> Self:
52
+ """Require an account reference exactly where ownership implies one."""
53
+ if self.kind == "account" and self.account_ref is None:
54
+ raise ValueError("account-owned data must name the owning account_ref")
55
+ if self.kind == "participant" and self.account_ref is not None:
56
+ raise ValueError(
57
+ "participant-owned data must not name an account_ref; use the "
58
+ "shared owner kind when an account also holds rights"
59
+ )
60
+ return self
61
+
62
+
63
+ class RawEpisodePolicy(ProtocolModel):
64
+ """What may happen to raw episode transcripts."""
65
+
66
+ local_retention: Literal["allowed", "prohibited", "required"]
67
+ server_upload: Permission
68
+ public_release: Permission
69
+ reproduction_access: Permission
70
+ training_use: Permission
71
+
72
+
73
+ class DerivedArtifactPolicy(ProtocolModel):
74
+ """What may happen to everything computed from the episodes."""
75
+
76
+ aggregate_scores: Visibility
77
+ uplift_report: Visibility
78
+ redacted_trace_projection: Visibility
79
+ anonymized_product_analytics: Permission
80
+
81
+
82
+ class CandidateSkillPolicy(ProtocolModel):
83
+ """Who owns the submitted skill and whether it can be published."""
84
+
85
+ ownership: Literal["participant", "account", "shared"]
86
+ public_release: Literal[
87
+ "required_for_climb",
88
+ "allowed",
89
+ "prohibited",
90
+ "consent_required",
91
+ ]
92
+ training_use: Permission
93
+
94
+
95
+ class RevocationPolicy(ProtocolModel):
96
+ """What a participant can withdraw later, and what stays published."""
97
+
98
+ future_use_revocable: bool
99
+ immutable_published_proofs_remain: bool
100
+
101
+
102
+ class DataPolicy(ProtocolModel):
103
+ """The complete rights statement a Campaign runs under."""
104
+
105
+ schema_version: Literal["techtree.data-policy.v1alpha1"]
106
+ id: NonEmptyString
107
+ version: int = Field(ge=1)
108
+ owner: DataOwner
109
+ raw_episodes: RawEpisodePolicy
110
+ derived_artifacts: DerivedArtifactPolicy
111
+ candidate_skill: CandidateSkillPolicy
112
+ revocation: RevocationPolicy
113
+
114
+ @model_validator(mode="after")
115
+ def _check_internal_consistency(self) -> Self:
116
+ """Reject a policy that permits a use it also makes impossible."""
117
+ if self.raw_episodes.local_retention == "prohibited":
118
+ for name in (
119
+ "server_upload",
120
+ "public_release",
121
+ "reproduction_access",
122
+ "training_use",
123
+ ):
124
+ if getattr(self.raw_episodes, name) != "prohibited":
125
+ raise ValueError(
126
+ f"raw_episodes.{name} cannot be permitted while "
127
+ "local_retention is prohibited; there would be nothing "
128
+ "left to share"
129
+ )
130
+ return self
@@ -0,0 +1,156 @@
1
+ """The managed Verifiers engine. Spec 11.13, decisions 0003 A8/A9.
2
+
3
+ The engine is a locked, content-addressed bundle: a Python version, a pinned
4
+ Verifiers revision, and the reference packages, described by an
5
+ ``EngineDescriptor``. The bundle's digest covers the static files and is not
6
+ stored inside the descriptor, because a document cannot contain its own hash.
7
+
8
+ Host platforms use one vocabulary — Go/OCI-style ``<os>/<arch>`` — everywhere
9
+ they appear. :func:`normalize_host_platform` is the only way a raw
10
+ ``sys.platform`` and ``platform.machine()`` pair becomes one of those strings,
11
+ and it refuses rather than guesses on anything else. An unsupported host is a
12
+ prerequisite failure with a name, not an install that fails later for reasons
13
+ the user cannot read.
14
+
15
+ The engine's host platform and the subject runtime's Docker platform share this
16
+ vocabulary but remain separate concepts: one is where Techtree runs, the other
17
+ is where the evaluated agent runs, and they are frequently not the same
18
+ machine.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ from typing import Final, Literal, Self
24
+
25
+ from pydantic import model_validator
26
+
27
+ from techtree.errors import PrerequisiteError
28
+ from techtree.models.base import (
29
+ Digest,
30
+ NonEmptyString,
31
+ ProtocolModel,
32
+ StateModel,
33
+ UtcDateTime,
34
+ )
35
+
36
+ __all__ = [
37
+ "EngineDescriptor",
38
+ "EngineInstallation",
39
+ "EnginePackage",
40
+ "EngineStatus",
41
+ "HostPlatform",
42
+ "normalize_host_platform",
43
+ ]
44
+
45
+
46
+ type HostPlatform = Literal[
47
+ "darwin/amd64",
48
+ "darwin/arm64",
49
+ "linux/amd64",
50
+ "linux/arm64",
51
+ ]
52
+ """The complete host-platform vocabulary. Decisions document 0003 A9."""
53
+
54
+ #: Every machine string that means the same architecture. The keys are what
55
+ #: ``platform.machine()`` reports across the platforms Techtree supports.
56
+ _ARCHITECTURES: Final[dict[str, str]] = {
57
+ "aarch64": "arm64",
58
+ "amd64": "amd64",
59
+ "arm64": "arm64",
60
+ "x86_64": "amd64",
61
+ }
62
+
63
+ #: The closed vocabulary, keyed by the pair it is derived from. A table rather
64
+ #: than string concatenation, so that the only values this function can return
65
+ #: are the four the protocol defines.
66
+ _HOST_PLATFORMS: Final[dict[tuple[str, str], HostPlatform]] = {
67
+ ("darwin", "amd64"): "darwin/amd64",
68
+ ("darwin", "arm64"): "darwin/arm64",
69
+ ("linux", "amd64"): "linux/amd64",
70
+ ("linux", "arm64"): "linux/arm64",
71
+ }
72
+
73
+
74
+ def normalize_host_platform(sys_platform: str, machine: str) -> HostPlatform:
75
+ """Return the ``<os>/<arch>`` name for a host, or refuse.
76
+
77
+ Raises :class:`~techtree.errors.PrerequisiteError` for any combination
78
+ outside the supported vocabulary. Guessing would produce an engine install
79
+ that resolves wheels for the wrong architecture and fails much later,
80
+ somewhere far less legible than here.
81
+ """
82
+ operating_system = sys_platform.strip().lower()
83
+ architecture = _ARCHITECTURES.get(machine.strip().lower(), "")
84
+ normalized = _HOST_PLATFORMS.get((operating_system, architecture))
85
+
86
+ if normalized is None:
87
+ raise PrerequisiteError(
88
+ f"unsupported host platform {sys_platform}/{machine}; Techtree "
89
+ "supports darwin and linux on arm64 and amd64",
90
+ code="unsupported_host_platform",
91
+ details={"sys_platform": sys_platform, "machine": machine},
92
+ )
93
+ return normalized
94
+
95
+
96
+ class EnginePackage(ProtocolModel):
97
+ """One package shipped inside the engine bundle."""
98
+
99
+ name: NonEmptyString
100
+ version: NonEmptyString
101
+ source_digest: Digest
102
+
103
+
104
+ class EngineDescriptor(ProtocolModel):
105
+ """What one engine bundle is, without saying what it hashes to."""
106
+
107
+ schema_version: Literal["techtree.engine.v1alpha1"]
108
+ name: NonEmptyString
109
+ python_version: NonEmptyString
110
+ verifiers_version: NonEmptyString
111
+ verifiers_revision: NonEmptyString
112
+ supported_hosts: list[HostPlatform]
113
+ packages: list[EnginePackage]
114
+
115
+ @model_validator(mode="after")
116
+ def _check_hosts_and_packages_are_listed_once(self) -> Self:
117
+ """Reject an empty or repeating host or package list."""
118
+ if not self.supported_hosts:
119
+ raise ValueError("an engine must support at least one host platform")
120
+ if len(set(self.supported_hosts)) != len(self.supported_hosts):
121
+ raise ValueError("supported_hosts must not repeat a platform")
122
+ names = [package.name for package in self.packages]
123
+ if len(set(names)) != len(names):
124
+ raise ValueError("an engine ships each package exactly once")
125
+ return self
126
+
127
+
128
+ class EngineInstallation(StateModel):
129
+ """A locally installed engine, as the registry records it."""
130
+
131
+ digest: Digest
132
+ installed_at: UtcDateTime
133
+ python_executable: NonEmptyString
134
+ descriptor_digest: Digest
135
+ verified: bool
136
+
137
+
138
+ class EngineStatus(ProtocolModel):
139
+ """What the CLI reports about one engine."""
140
+
141
+ digest: Digest
142
+ installed: bool
143
+ active: bool
144
+ verified: bool
145
+ path: NonEmptyString
146
+ python_executable: NonEmptyString | None
147
+ detail: NonEmptyString
148
+
149
+ @model_validator(mode="after")
150
+ def _check_status_is_coherent(self) -> Self:
151
+ """Reject a status that verifies or activates an engine that is absent."""
152
+ if not self.installed and (self.verified or self.active):
153
+ raise ValueError(
154
+ "an engine that is not installed cannot be active or verified"
155
+ )
156
+ return self
@@ -0,0 +1,130 @@
1
+ """What one episode produced. Spec section 11.10.
2
+
3
+ The shape is frozen now and fake-populated until WP6, which is the point: the
4
+ statuses that say "this number is not evidence" have to exist before the first
5
+ number does. ``development_only`` is a first-class score and evidence status
6
+ rather than an absent field, so a fake receipt is unmistakably fake to a reader
7
+ and to a service, and nothing has to infer it from context.
8
+
9
+ A receipt points at ``campaign_spec_digest`` and carries the data policy that
10
+ governed it. The public Climb is an optional context, never the anchor: a
11
+ reproduction run has no Climb, and its receipts must still be complete.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ from enum import StrEnum
17
+ from typing import Literal, Self
18
+
19
+ from pydantic import model_validator
20
+
21
+ from techtree.models.base import (
22
+ ArtifactRef,
23
+ Digest,
24
+ NonEmptyString,
25
+ ProtocolModel,
26
+ )
27
+ from techtree.models.campaign import ProgramRef, PublicContext
28
+ from techtree.models.evaluation_backend import EvaluationBackendSpec
29
+ from techtree.models.experiment import ExperimentVariant
30
+
31
+ __all__ = [
32
+ "EpisodeReceipt",
33
+ "EvidenceStatus",
34
+ "NamedTraceReceipt",
35
+ "ScoreStatus",
36
+ "SubjectRuntimeReceipt",
37
+ ]
38
+
39
+
40
+ class ScoreStatus(StrEnum):
41
+ """How much weight the recorded reward carries."""
42
+
43
+ PENDING = "pending"
44
+ VALID = "valid"
45
+ INVALID = "invalid"
46
+ ERRORED = "errored"
47
+ MISSING = "missing"
48
+ DEVELOPMENT_ONLY = "development_only"
49
+
50
+
51
+ class EvidenceStatus(StrEnum):
52
+ """How complete the supporting evidence is."""
53
+
54
+ NOT_COLLECTED = "not_collected"
55
+ COMPLETE = "complete"
56
+ PARTIAL = "partial"
57
+ INVALID = "invalid"
58
+ DEVELOPMENT_ONLY = "development_only"
59
+
60
+
61
+ class NamedTraceReceipt(ProtocolModel):
62
+ """One named trace within an episode."""
63
+
64
+ role: NonEmptyString
65
+ trace_id: NonEmptyString
66
+ trace_digest: Digest
67
+ task_hash: Digest
68
+ rewards: dict[str, float]
69
+ metrics: dict[str, float | None]
70
+ ok: bool
71
+
72
+
73
+ class SubjectRuntimeReceipt(ProtocolModel):
74
+ """Where the subject agent actually executed, if it executed."""
75
+
76
+ kind: Literal["not_executed", "docker"]
77
+ resolved_image_digest: Digest | None = None
78
+ platform: NonEmptyString | None = None
79
+
80
+ @model_validator(mode="after")
81
+ def _check_runtime_evidence_matches_kind(self) -> Self:
82
+ """Reject runtime detail on an episode that never ran a runtime."""
83
+ if self.kind == "not_executed" and (
84
+ self.resolved_image_digest is not None or self.platform is not None
85
+ ):
86
+ raise ValueError(
87
+ "an episode that did not execute a runtime cannot report the "
88
+ "image or platform it executed on"
89
+ )
90
+ if self.kind == "docker" and self.resolved_image_digest is None:
91
+ raise ValueError("a docker episode records the image digest it ran")
92
+ return self
93
+
94
+
95
+ class EpisodeReceipt(ProtocolModel):
96
+ """The complete record of one scored episode."""
97
+
98
+ schema_version: Literal["techtree.episode-receipt.v1alpha1"]
99
+ id: NonEmptyString
100
+ run_id: NonEmptyString
101
+ campaign_spec_digest: Digest
102
+ program_ref: ProgramRef | None
103
+ public_context: PublicContext | None
104
+ data_policy_digest: Digest
105
+ outcome_contract_digest: Digest | None
106
+ evaluation_backend: EvaluationBackendSpec
107
+ subject_runtime: SubjectRuntimeReceipt
108
+ variant: ExperimentVariant
109
+ experiment_manifest_digest: Digest
110
+ episode_id: NonEmptyString
111
+ episode_digest: Digest
112
+ task_hash: Digest
113
+ named_traces: dict[str, list[NamedTraceReceipt]]
114
+ score_status: ScoreStatus
115
+ evidence_status: EvidenceStatus
116
+ execution_backend: Literal["fake", "verifiers"]
117
+ artifacts: list[ArtifactRef]
118
+
119
+ @model_validator(mode="after")
120
+ def _check_fake_episodes_are_unmistakably_fake(self) -> Self:
121
+ """Refuse to let a fake episode wear a real score."""
122
+ if self.execution_backend == "fake" and (
123
+ self.score_status is not ScoreStatus.DEVELOPMENT_ONLY
124
+ or self.evidence_status is not EvidenceStatus.DEVELOPMENT_ONLY
125
+ ):
126
+ raise ValueError(
127
+ "a fake episode reports development_only score and evidence; "
128
+ "any other status would present invented numbers as results"
129
+ )
130
+ return self
@@ -0,0 +1,113 @@
1
+ """Who orchestrated and attested to an evaluation. Spec section 11.3.
2
+
3
+ The evaluation backend and the subject runtime answer two different questions
4
+ and are deliberately separate objects:
5
+
6
+ ``EvaluationBackendSpec``
7
+ Who ran the comparison and whose word the result rests on.
8
+
9
+ ``RuntimeSpec`` (in :mod:`techtree.models.campaign`)
10
+ Where the evaluated agent's process actually executed.
11
+
12
+ Conflating them is how a self-reported local result quietly acquires the
13
+ authority of a platform-attested one. Each backend kind therefore fixes the
14
+ attestation it is allowed to claim, and the references that must or must not
15
+ accompany it.
16
+
17
+ The enum keeps the future backends so that stored documents and published
18
+ schemas do not need a version bump when they arrive. Services enforce the
19
+ narrower WP0–WP5 rule — only ``local_techtree`` — at the point of use, not
20
+ here, because a document that merely mentions a future backend must still be
21
+ parseable.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ from enum import StrEnum
27
+ from typing import Literal, Self
28
+
29
+ from pydantic import model_validator
30
+
31
+ from techtree.models.base import NonEmptyString, ProtocolModel
32
+
33
+ __all__ = [
34
+ "SUPPORTED_EVALUATION_BACKEND_KINDS",
35
+ "AttestationKind",
36
+ "EvaluationBackendKind",
37
+ "EvaluationBackendSpec",
38
+ ]
39
+
40
+
41
+ class EvaluationBackendKind(StrEnum):
42
+ """Which system orchestrated the evaluation."""
43
+
44
+ LOCAL_TECHTREE = "local_techtree"
45
+ PRIME_LAB = "prime_lab"
46
+ INDEPENDENT_REPRODUCER = "independent_reproducer"
47
+
48
+
49
+ class AttestationKind(StrEnum):
50
+ """Whose attestation the recorded result carries."""
51
+
52
+ PARTICIPANT = "participant"
53
+ PLATFORM = "platform"
54
+ INDEPENDENT = "independent"
55
+
56
+
57
+ #: The only backend kinds WP0–WP5 services accept. The schema is wider than
58
+ #: this on purpose; the runtime surface is not.
59
+ SUPPORTED_EVALUATION_BACKEND_KINDS: frozenset[EvaluationBackendKind] = frozenset(
60
+ {EvaluationBackendKind.LOCAL_TECHTREE}
61
+ )
62
+
63
+
64
+ class EvaluationBackendSpec(ProtocolModel):
65
+ """The orchestrating backend and the attestation it carries."""
66
+
67
+ schema_version: Literal["techtree.evaluation-backend.v1alpha1"]
68
+ kind: EvaluationBackendKind
69
+ attestation: AttestationKind
70
+ workspace_ref: NonEmptyString | None = None
71
+ provider_run_ref: NonEmptyString | None = None
72
+ executor_identity: NonEmptyString | None = None
73
+
74
+ @model_validator(mode="after")
75
+ def _check_kind_agrees_with_its_evidence(self) -> Self:
76
+ """Reject attestation and reference combinations a kind cannot support."""
77
+ if self.kind is EvaluationBackendKind.LOCAL_TECHTREE:
78
+ if self.attestation is not AttestationKind.PARTICIPANT:
79
+ raise ValueError(
80
+ "local_techtree evaluation is self-reported, so its "
81
+ "attestation must be participant"
82
+ )
83
+ if self.workspace_ref is not None:
84
+ raise ValueError("local_techtree evaluation has no workspace_ref")
85
+ if self.provider_run_ref is not None:
86
+ raise ValueError("local_techtree evaluation has no provider_run_ref")
87
+ # executor_identity stays optional: WP0-WP5 record no identity.
88
+ return self
89
+
90
+ if self.kind is EvaluationBackendKind.PRIME_LAB:
91
+ if self.attestation is not AttestationKind.PLATFORM:
92
+ raise ValueError(
93
+ "prime_lab evaluation is platform-attested, so its "
94
+ "attestation must be platform"
95
+ )
96
+ if self.workspace_ref is None and self.provider_run_ref is None:
97
+ raise ValueError(
98
+ "prime_lab evaluation must carry a workspace_ref or a "
99
+ "provider_run_ref so the platform record can be found"
100
+ )
101
+ return self
102
+
103
+ if self.attestation is not AttestationKind.INDEPENDENT:
104
+ raise ValueError(
105
+ "independent_reproducer evaluation must carry an independent "
106
+ "attestation"
107
+ )
108
+ if self.executor_identity is None:
109
+ raise ValueError(
110
+ "independent_reproducer evaluation must name the executor_identity "
111
+ "that stands behind it"
112
+ )
113
+ return self
@@ -0,0 +1,154 @@
1
+ """One resolved variant of a Campaign, and the comparison. Spec section 11.8.
2
+
3
+ An ``ExperimentManifest`` is a Campaign with every choice made: this taskset,
4
+ this agent, this skill list. Two of them — baseline and candidate — are what a
5
+ run actually executes.
6
+
7
+ The comparison deliberately looks at ``configuration`` and nothing else. A
8
+ manifest identifier, a variant name, a creation time, and a public context all
9
+ differ between two correct manifests of the same experiment; comparing them
10
+ would report differences that mean nothing and bury the one difference that
11
+ means everything. ``ManifestComparison`` therefore compares the configurations
12
+ and reports whether the only difference found is the one the mutation contract
13
+ permits.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from enum import StrEnum
19
+ from typing import Literal, Self
20
+
21
+ from pydantic import model_validator
22
+
23
+ from techtree.models.base import (
24
+ Digest,
25
+ JsonValue,
26
+ NonEmptyString,
27
+ ProtocolModel,
28
+ UtcDateTime,
29
+ )
30
+ from techtree.models.campaign import (
31
+ AgentSpec,
32
+ BudgetSpec,
33
+ CampaignTaskset,
34
+ EnvironmentSpec,
35
+ EvidenceRequirements,
36
+ ExecutionSpec,
37
+ MutationContract,
38
+ MutationKind,
39
+ ProgramRef,
40
+ PublicContext,
41
+ ScoringSpec,
42
+ )
43
+ from techtree.models.evaluation_backend import EvaluationBackendSpec
44
+
45
+ __all__ = [
46
+ "ExperimentConfiguration",
47
+ "ExperimentManifest",
48
+ "ExperimentVariant",
49
+ "JsonDifference",
50
+ "ManifestComparison",
51
+ ]
52
+
53
+
54
+ class ExperimentVariant(StrEnum):
55
+ """Which side of the comparison a manifest describes."""
56
+
57
+ BASELINE = "baseline"
58
+ CANDIDATE = "candidate"
59
+
60
+
61
+ class ExperimentConfiguration(ProtocolModel):
62
+ """The part of a manifest that is compared.
63
+
64
+ This mirrors the Campaign's scientific fields with the choices resolved. It
65
+ holds no identifier, no timestamp, and no public context, precisely so that
66
+ two manifests of the same experiment have byte-identical configurations
67
+ except where the mutation contract allows them to differ.
68
+ """
69
+
70
+ taskset: CampaignTaskset
71
+ environment: EnvironmentSpec
72
+ agents: dict[str, AgentSpec]
73
+ mutation_contract: MutationContract
74
+ evaluation_backend: EvaluationBackendSpec
75
+ execution: ExecutionSpec
76
+ scoring: ScoringSpec
77
+ evidence: EvidenceRequirements
78
+ budgets: BudgetSpec
79
+ data_policy_digest: Digest
80
+ outcome_contract_digest: Digest | None
81
+
82
+
83
+ class ExperimentManifest(ProtocolModel):
84
+ """One fully resolved variant derived from a Campaign."""
85
+
86
+ schema_version: Literal["techtree.experiment.v1alpha1"]
87
+ id: NonEmptyString
88
+ campaign_spec_digest: Digest
89
+ program_ref: ProgramRef | None
90
+ public_context: PublicContext | None
91
+ variant: ExperimentVariant
92
+ configuration: ExperimentConfiguration
93
+ configuration_digest: Digest
94
+ created_at: UtcDateTime
95
+
96
+ @model_validator(mode="after")
97
+ def _check_variant_skill_count(self) -> Self:
98
+ """Hold each variant to the skill count its mutation kind requires."""
99
+ subject = self.configuration.agents.get("subject")
100
+ if subject is None:
101
+ raise ValueError("an experiment configuration defines a subject agent")
102
+ skills = len(subject.harness.skills)
103
+
104
+ if self.variant is ExperimentVariant.CANDIDATE:
105
+ if skills != 1:
106
+ raise ValueError("the candidate variant carries exactly one skill")
107
+ return self
108
+
109
+ # What the baseline carries is what the candidate is measured against,
110
+ # and the mutation kind is what says which that is (spec section 3.1):
111
+ # nothing for an insertion, the skill being revised for a replacement.
112
+ # The two sides' root digests are a property of the pair rather than of
113
+ # one manifest, so they are checked where the pair is — the builder when
114
+ # one is derived, ``manifests.compare`` when two are read back.
115
+ if self.configuration.mutation_contract.kind is MutationKind.SKILL_INSERTION:
116
+ if skills != 0:
117
+ raise ValueError("the baseline variant carries no candidate skill")
118
+ elif skills != 1:
119
+ raise ValueError(
120
+ "the baseline variant of a skill_replacement carries exactly one "
121
+ "skill to replace"
122
+ )
123
+ return self
124
+
125
+
126
+ class JsonDifference(ProtocolModel):
127
+ """One JSON Pointer at which two configurations disagree."""
128
+
129
+ pointer: NonEmptyString
130
+ baseline: JsonValue | None
131
+ candidate: JsonValue | None
132
+
133
+
134
+ class ManifestComparison(ProtocolModel):
135
+ """Whether the candidate differs from the baseline only where permitted."""
136
+
137
+ baseline_configuration_digest: Digest
138
+ candidate_configuration_digest: Digest
139
+ differences: list[JsonDifference]
140
+ allowed_differences: list[NonEmptyString]
141
+ controlled: bool
142
+ violations: list[NonEmptyString]
143
+
144
+ @model_validator(mode="after")
145
+ def _check_control_agrees_with_violations(self) -> Self:
146
+ """Reject a comparison that calls itself controlled while listing faults."""
147
+ if self.controlled and self.violations:
148
+ raise ValueError(
149
+ "a controlled comparison has no violations; listing both leaves "
150
+ "a reader unable to tell which one is true"
151
+ )
152
+ if not self.controlled and not self.violations:
153
+ raise ValueError("an uncontrolled comparison must say which rule it broke")
154
+ return self