techtree 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- techtree/__init__.py +35 -0
- techtree/__main__.py +14 -0
- techtree/canonical.py +239 -0
- techtree/catalog/__init__.py +25 -0
- techtree/catalog/repository.py +400 -0
- techtree/catalog/service.py +419 -0
- techtree/cli/__init__.py +1 -0
- techtree/cli/app.py +416 -0
- techtree/cli/commands/__init__.py +1 -0
- techtree/cli/commands/climb.py +1223 -0
- techtree/cli/commands/doctor.py +147 -0
- techtree/cli/commands/engine.py +207 -0
- techtree/cli/commands/proof.py +556 -0
- techtree/cli/commands/publish.py +447 -0
- techtree/cli/commands/release.py +303 -0
- techtree/cli/commands/run.py +1067 -0
- techtree/cli/commands/setup.py +181 -0
- techtree/cli/commands/skill.py +221 -0
- techtree/cli/commands/uplift.py +698 -0
- techtree/cli/commands/withdraw.py +212 -0
- techtree/cli/confirm.py +47 -0
- techtree/cli/context.py +96 -0
- techtree/cli/invoke.py +220 -0
- techtree/cli/output.py +280 -0
- techtree/constants.py +138 -0
- techtree/crypto.py +128 -0
- techtree/doctor/__init__.py +1 -0
- techtree/doctor/checks.py +675 -0
- techtree/doctor/execution_checks.py +435 -0
- techtree/doctor/service.py +326 -0
- techtree/drafts/__init__.py +32 -0
- techtree/drafts/source.py +146 -0
- techtree/drafts/store.py +992 -0
- techtree/engines/__init__.py +1 -0
- techtree/engines/bundle.py +251 -0
- techtree/engines/installer.py +679 -0
- techtree/engines/registry.py +235 -0
- techtree/engines/runner.py +170 -0
- techtree/errors.py +262 -0
- techtree/fs.py +234 -0
- techtree/harness.py +108 -0
- techtree/identity/__init__.py +41 -0
- techtree/identity/models.py +113 -0
- techtree/identity/service.py +199 -0
- techtree/identity/store.py +263 -0
- techtree/ids.py +85 -0
- techtree/manifests/__init__.py +39 -0
- techtree/manifests/builder.py +433 -0
- techtree/manifests/compare.py +376 -0
- techtree/models/__init__.py +282 -0
- techtree/models/base.py +201 -0
- techtree/models/campaign.py +484 -0
- techtree/models/catalog.py +227 -0
- techtree/models/cli.py +151 -0
- techtree/models/climb.py +254 -0
- techtree/models/data_policy.py +130 -0
- techtree/models/engine.py +156 -0
- techtree/models/episode_receipt.py +130 -0
- techtree/models/evaluation_backend.py +113 -0
- techtree/models/experiment.py +154 -0
- techtree/models/run.py +214 -0
- techtree/models/skill.py +156 -0
- techtree/models/uplift_report.py +158 -0
- techtree/models/validation.py +299 -0
- techtree/paths.py +116 -0
- techtree/presentation/__init__.py +31 -0
- techtree/presentation/build.py +1242 -0
- techtree/presentation/compact.py +246 -0
- techtree/presentation/evidence.py +169 -0
- techtree/presentation/models.py +358 -0
- techtree/presentation/rich.py +312 -0
- techtree/presentation/sanitize.py +156 -0
- techtree/publication/__init__.py +44 -0
- techtree/publication/address.py +180 -0
- techtree/publication/coordinates.py +26 -0
- techtree/publication/journal.py +212 -0
- techtree/publication/keccak.py +183 -0
- techtree/publication/models.py +209 -0
- techtree/publication/offer.py +35 -0
- techtree/publication/service.py +618 -0
- techtree/publication/transport.py +296 -0
- techtree/publication/verify.py +242 -0
- techtree/publication/withdraw.py +156 -0
- techtree/py.typed +0 -0
- techtree/receipts/__init__.py +52 -0
- techtree/receipts/bundle.py +578 -0
- techtree/receipts/compare.py +1065 -0
- techtree/receipts/episode.py +672 -0
- techtree/receipts/execution.py +630 -0
- techtree/receipts/observed.py +474 -0
- techtree/receipts/set.py +336 -0
- techtree/receipts/uplift.py +655 -0
- techtree/receipts/verify.py +1055 -0
- techtree/release/__init__.py +9 -0
- techtree/release/bootstrap.py +509 -0
- techtree/release/checks.py +376 -0
- techtree/release/document.py +125 -0
- techtree/release/generate.py +221 -0
- techtree/release/models.py +293 -0
- techtree/release/provenance.py +109 -0
- techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
- techtree/resources/catalog/catalog.json +32 -0
- techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
- techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
- techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
- techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
- techtree/resources/engines/default/engine.json +20 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
- techtree/resources/engines/default/pyproject.toml +23 -0
- techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
- techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
- techtree/resources/engines/default/tools/normalize_validation.py +222 -0
- techtree/resources/engines/default/uv.lock +1758 -0
- techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
- techtree/resources/release/build-provenance.json +4 -0
- techtree/resources/release/release-core.json +24 -0
- techtree/runs/__init__.py +31 -0
- techtree/runs/artifacts.py +750 -0
- techtree/runs/child_registry.py +228 -0
- techtree/runs/events.py +478 -0
- techtree/runs/executor.py +140 -0
- techtree/runs/fake.py +741 -0
- techtree/runs/launcher.py +253 -0
- techtree/runs/machine.py +489 -0
- techtree/runs/real.py +789 -0
- techtree/runs/service.py +616 -0
- techtree/runs/store.py +555 -0
- techtree/runs/validation.py +259 -0
- techtree/runs/variants.py +684 -0
- techtree/settings.py +143 -0
- techtree/skills/__init__.py +14 -0
- techtree/skills/archive.py +282 -0
- techtree/skills/policy.py +62 -0
- techtree/skills/scanner.py +394 -0
- techtree/skills/service.py +752 -0
- techtree/skills/starter.py +434 -0
- techtree/tasksets/__init__.py +1 -0
- techtree/tasksets/membership.py +269 -0
- techtree/tasksets/provider.py +207 -0
- techtree/tasksets/resolver.py +311 -0
- techtree/tasksets/service.py +484 -0
- techtree/tasksets/verifiers_cli.py +538 -0
- techtree/uplift/__init__.py +20 -0
- techtree/uplift/context.py +544 -0
- techtree/uplift/derive.py +203 -0
- techtree/uplift/public_tasks.py +151 -0
- techtree/uplift/service.py +719 -0
- techtree/uplift/source.py +160 -0
- techtree/verifiers/__init__.py +31 -0
- techtree/verifiers/budget.py +219 -0
- techtree/verifiers/child.py +633 -0
- techtree/verifiers/compiler.py +432 -0
- techtree/verifiers/config.py +365 -0
- techtree/verifiers/credentials.py +321 -0
- techtree/verifiers/image.py +126 -0
- techtree/verifiers/models.py +527 -0
- techtree/verifiers/outputs.py +368 -0
- techtree/verifiers/progress.py +192 -0
- techtree/verifiers/supervisor.py +341 -0
- techtree/verifiers/verify.py +782 -0
- techtree/version.py +39 -0
- techtree/worker/__init__.py +18 -0
- techtree/worker/execute.py +487 -0
- techtree/worker/main.py +57 -0
- techtree-0.1.0.dist-info/METADATA +344 -0
- techtree-0.1.0.dist-info/RECORD +174 -0
- techtree-0.1.0.dist-info/WHEEL +4 -0
- techtree-0.1.0.dist-info/entry_points.txt +3 -0
- techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
techtree/models/run.py
ADDED
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
"""Run requests, phases, events, and local state. Spec 11.12, decisions 0003.
|
|
2
|
+
|
|
3
|
+
A run has one phase at a time and moves forward through them. The phase names
|
|
4
|
+
are part of the CLI contract, so they are an enum rather than free text, and
|
|
5
|
+
the event stream records both the phase entered and the phase left — a reader
|
|
6
|
+
reconstructing a run from events should never have to infer where it came from.
|
|
7
|
+
|
|
8
|
+
``RunRequest`` is immutable: it is what was asked for. ``RunState`` is a
|
|
9
|
+
``StateModel`` because it is what is currently true, rewritten as the worker
|
|
10
|
+
makes progress. Keeping them in separate classes is what stops a heartbeat
|
|
11
|
+
update from being able to alter the request it is executing.
|
|
12
|
+
|
|
13
|
+
Decisions document 0003 A5 puts ``policy_acknowledgement`` here. A draft states
|
|
14
|
+
which rights policy must be accepted; the run records that it *was* accepted,
|
|
15
|
+
by which method, and when. Decisions document 0019 section 2 makes that one
|
|
16
|
+
answer to one review rather than a handle that had to be presented: nothing is
|
|
17
|
+
started without the review having been shown and explicitly accepted, and a
|
|
18
|
+
caller that cannot be asked has to say so with a flag.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
from enum import StrEnum
|
|
24
|
+
from typing import Literal, Self
|
|
25
|
+
|
|
26
|
+
from pydantic import Field, model_validator
|
|
27
|
+
|
|
28
|
+
from techtree.models.base import (
|
|
29
|
+
Digest,
|
|
30
|
+
JsonValue,
|
|
31
|
+
NonEmptyString,
|
|
32
|
+
ProtocolModel,
|
|
33
|
+
StateModel,
|
|
34
|
+
UtcDateTime,
|
|
35
|
+
)
|
|
36
|
+
from techtree.models.campaign import ProgramRef, PublicContext
|
|
37
|
+
from techtree.models.cli import CliError
|
|
38
|
+
from techtree.models.evaluation_backend import EvaluationBackendSpec
|
|
39
|
+
|
|
40
|
+
__all__ = [
|
|
41
|
+
"ExecutorKind",
|
|
42
|
+
"PolicyAcknowledgement",
|
|
43
|
+
"RunEvent",
|
|
44
|
+
"RunPhase",
|
|
45
|
+
"RunProgress",
|
|
46
|
+
"RunRequest",
|
|
47
|
+
"RunState",
|
|
48
|
+
"RunStatus",
|
|
49
|
+
"VariantProgress",
|
|
50
|
+
]
|
|
51
|
+
|
|
52
|
+
type ExecutorKind = Literal["fake", "verifiers"]
|
|
53
|
+
"""Which executor a run was created to be executed by.
|
|
54
|
+
|
|
55
|
+
The two names are the two that exist, and they are the same two an
|
|
56
|
+
:class:`~techtree.models.episode_receipt.EpisodeReceipt` records under
|
|
57
|
+
``execution_backend`` — a run and the receipts it produces should not need a
|
|
58
|
+
translation table to agree on what ran.
|
|
59
|
+
|
|
60
|
+
The value is decided when the run is created, from the Campaign, because that
|
|
61
|
+
is when a person is told what is about to happen and what it will cost. It is
|
|
62
|
+
a record of the answer, not the place the answer is worked out: the worker
|
|
63
|
+
re-reads the run's own staged Campaign and refuses a request that disagrees
|
|
64
|
+
with it, so a hand-edited request buys nothing.
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class RunPhase(StrEnum):
|
|
69
|
+
"""Where a run currently is."""
|
|
70
|
+
|
|
71
|
+
CREATED = "created"
|
|
72
|
+
VALIDATING_TASKSET = "validating_taskset"
|
|
73
|
+
RUNNING_BASELINE = "running_baseline"
|
|
74
|
+
RUNNING_CANDIDATE = "running_candidate"
|
|
75
|
+
#: Both variants in flight at once. The sequential pair above stays for the
|
|
76
|
+
#: fake executor; spec section 3.3 adds this one rather than replacing them.
|
|
77
|
+
RUNNING_VARIANTS = "running_variants"
|
|
78
|
+
BUILDING_RECEIPTS = "building_receipts"
|
|
79
|
+
VERIFYING_COMPARISON = "verifying_comparison"
|
|
80
|
+
BUILDING_REPORT = "building_report"
|
|
81
|
+
COMPLETED = "completed"
|
|
82
|
+
FAILED = "failed"
|
|
83
|
+
CANCEL_REQUESTED = "cancel_requested"
|
|
84
|
+
CANCELLED = "cancelled"
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class PolicyAcknowledgement(ProtocolModel):
|
|
88
|
+
"""That a specific rights policy was accepted, how, and when.
|
|
89
|
+
|
|
90
|
+
Decisions document 0003 A5, as amended by 0019 section 2.
|
|
91
|
+
``explicit_cli_review`` says the review of what the run would do — its
|
|
92
|
+
size, the spending limit its Campaign declares, the one change being
|
|
93
|
+
measured, where model calls go, and what an upload would carry — was put in
|
|
94
|
+
front of whoever started it and
|
|
95
|
+
explicitly accepted. It covers both spellings of that answer at the command
|
|
96
|
+
line, the typed ``y`` and the flag an operator passes instead, because the
|
|
97
|
+
fact being recorded is the same one; who gave the answer is recorded
|
|
98
|
+
separately on the run's ``run.approved`` event.
|
|
99
|
+
|
|
100
|
+
``host_agent_confirmation`` is the approval surface the plugin presents in
|
|
101
|
+
a conversation: the review is shown there, the person confirms there, and
|
|
102
|
+
the plugin then starts that exact draft. The process that writes this
|
|
103
|
+
record is not the one that asked the question, so which surface it was is
|
|
104
|
+
declared by whoever starts the run rather than guessed at from the fact
|
|
105
|
+
that nobody was prompted here.
|
|
106
|
+
"""
|
|
107
|
+
|
|
108
|
+
data_policy_digest: Digest
|
|
109
|
+
method: Literal[
|
|
110
|
+
"explicit_cli_review",
|
|
111
|
+
"host_agent_confirmation",
|
|
112
|
+
]
|
|
113
|
+
acknowledged_at: UtcDateTime
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
class RunRequest(ProtocolModel):
|
|
117
|
+
"""What was asked for, fixed at the moment the run was created."""
|
|
118
|
+
|
|
119
|
+
run_id: NonEmptyString
|
|
120
|
+
draft_id: NonEmptyString
|
|
121
|
+
draft_digest: Digest
|
|
122
|
+
campaign_spec_digest: Digest
|
|
123
|
+
program_ref: ProgramRef | None
|
|
124
|
+
public_context: PublicContext | None
|
|
125
|
+
data_policy_digest: Digest
|
|
126
|
+
outcome_contract_digest: Digest | None
|
|
127
|
+
evaluation_backend: EvaluationBackendSpec
|
|
128
|
+
taskset_lock_digest: Digest | None
|
|
129
|
+
baseline_manifest_digest: Digest
|
|
130
|
+
candidate_manifest_digest: Digest
|
|
131
|
+
policy_acknowledgement: PolicyAcknowledgement
|
|
132
|
+
executor_kind: ExecutorKind
|
|
133
|
+
created_at: UtcDateTime
|
|
134
|
+
|
|
135
|
+
@model_validator(mode="after")
|
|
136
|
+
def _check_acknowledged_policy_is_the_one_being_run(self) -> Self:
|
|
137
|
+
"""Reject a run acknowledging a different policy than it executes under."""
|
|
138
|
+
if self.policy_acknowledgement.data_policy_digest != self.data_policy_digest:
|
|
139
|
+
raise ValueError(
|
|
140
|
+
"the acknowledged DataPolicy is not the one this run executes under"
|
|
141
|
+
)
|
|
142
|
+
if self.baseline_manifest_digest == self.candidate_manifest_digest:
|
|
143
|
+
raise ValueError("a run compares two different manifests")
|
|
144
|
+
return self
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
class RunEvent(ProtocolModel):
|
|
148
|
+
"""One appended record of something that happened to a run."""
|
|
149
|
+
|
|
150
|
+
sequence: int = Field(ge=0)
|
|
151
|
+
timestamp: UtcDateTime
|
|
152
|
+
run_id: NonEmptyString
|
|
153
|
+
previous_phase: RunPhase | None
|
|
154
|
+
phase: RunPhase
|
|
155
|
+
kind: NonEmptyString
|
|
156
|
+
details: dict[str, JsonValue]
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
class RunProgress(StateModel):
|
|
160
|
+
"""How far through the current phase the worker is."""
|
|
161
|
+
|
|
162
|
+
current: int = Field(ge=0)
|
|
163
|
+
total: int = Field(ge=0)
|
|
164
|
+
label: NonEmptyString
|
|
165
|
+
|
|
166
|
+
@model_validator(mode="after")
|
|
167
|
+
def _check_progress_is_within_its_total(self) -> Self:
|
|
168
|
+
"""Reject progress that has passed the end of the work it describes."""
|
|
169
|
+
if self.current > self.total:
|
|
170
|
+
raise ValueError("progress cannot exceed its total")
|
|
171
|
+
return self
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
class VariantProgress(StateModel):
|
|
175
|
+
"""How far one side of a concurrent comparison has got. Spec section 3.3.
|
|
176
|
+
|
|
177
|
+
``RunProgress`` measures one position in one phase, which is all a
|
|
178
|
+
sequential run has to report. When both variants are in flight there are two
|
|
179
|
+
positions at once, and each carries its own episode counts and its own
|
|
180
|
+
lifecycle, so they are projected side by side rather than flattened.
|
|
181
|
+
"""
|
|
182
|
+
|
|
183
|
+
variant: Literal["baseline", "candidate"]
|
|
184
|
+
completed: int = Field(ge=0)
|
|
185
|
+
total: int = Field(ge=0)
|
|
186
|
+
running: int = Field(ge=0)
|
|
187
|
+
errored: int = Field(ge=0)
|
|
188
|
+
state: Literal["pending", "running", "completed", "failed", "cancelled"]
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
class RunState(StateModel):
|
|
192
|
+
"""What is currently true of a run, rewritten as it advances."""
|
|
193
|
+
|
|
194
|
+
run_id: NonEmptyString
|
|
195
|
+
phase: RunPhase
|
|
196
|
+
sequence: int = Field(ge=0)
|
|
197
|
+
updated_at: UtcDateTime
|
|
198
|
+
worker_pid: int | None
|
|
199
|
+
worker_started_at: UtcDateTime | None
|
|
200
|
+
heartbeat_at: UtcDateTime | None
|
|
201
|
+
cancel_requested_at: UtcDateTime | None
|
|
202
|
+
error: CliError | None
|
|
203
|
+
progress: RunProgress | None
|
|
204
|
+
variant_progress: dict[str, VariantProgress] = Field(default_factory=dict)
|
|
205
|
+
result_digest: Digest | None
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
class RunStatus(ProtocolModel):
|
|
209
|
+
"""A run's state plus the liveness facts only the host can determine."""
|
|
210
|
+
|
|
211
|
+
state: RunState
|
|
212
|
+
worker_alive: bool
|
|
213
|
+
heartbeat_stale: bool
|
|
214
|
+
result_available: bool
|
techtree/models/skill.py
ADDED
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
"""Candidate skills and the draft that submits them. Spec 11.7, decisions 0003.
|
|
2
|
+
|
|
3
|
+
A ``SkillArtifact`` is the content-addressed description of what a participant
|
|
4
|
+
wrote: the files, their digests, and the digest of the tree as a whole. No
|
|
5
|
+
local path ever enters it, because a draft that recorded where a skill lived on
|
|
6
|
+
one machine would not be reproducible on another and would leak the author's
|
|
7
|
+
directory layout into a public object.
|
|
8
|
+
|
|
9
|
+
A ``SubmissionDraft`` is the complete, reviewable statement of what is about to
|
|
10
|
+
be run, including the rights the participant is being asked to accept.
|
|
11
|
+
Decisions document 0003 A5 splits that from the moment of accepting: the draft
|
|
12
|
+
carries a ``PolicyAcceptanceRequirement`` — the digest, whether acceptance is
|
|
13
|
+
required, and a stable human-readable summary — while the acknowledgement
|
|
14
|
+
itself is recorded on the ``RunRequest``. Stating what will have to be accepted
|
|
15
|
+
and recording that it was accepted are two different facts, and the two objects
|
|
16
|
+
keep them from being blurred.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from typing import Final, Literal, Self
|
|
22
|
+
|
|
23
|
+
from pydantic import Field, model_validator
|
|
24
|
+
|
|
25
|
+
from techtree.models.base import (
|
|
26
|
+
Digest,
|
|
27
|
+
NonEmptyString,
|
|
28
|
+
ProtocolModel,
|
|
29
|
+
UtcDateTime,
|
|
30
|
+
)
|
|
31
|
+
from techtree.models.campaign import ProgramRef, PublicContext
|
|
32
|
+
|
|
33
|
+
__all__ = [
|
|
34
|
+
"SKILL_ENTRY_FILE",
|
|
35
|
+
"PolicyAcceptanceRequirement",
|
|
36
|
+
"SkillArtifact",
|
|
37
|
+
"SkillFile",
|
|
38
|
+
"SubmissionDraft",
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
#: The one file every instruction skill must contain. Its absence is not a
|
|
42
|
+
#: warning: a skill with no instructions is not a skill.
|
|
43
|
+
SKILL_ENTRY_FILE: Final = "SKILL.md"
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _check_relative_posix_path(value: str) -> None:
|
|
47
|
+
"""Raise when a path is absolute, escaping, or not POSIX-spelled."""
|
|
48
|
+
if value != value.strip():
|
|
49
|
+
raise ValueError(f"skill path has surrounding whitespace: {value!r}")
|
|
50
|
+
if value.startswith("/") or (len(value) > 1 and value[1] == ":"):
|
|
51
|
+
raise ValueError(f"skill paths are relative; got {value!r}")
|
|
52
|
+
if "\\" in value:
|
|
53
|
+
raise ValueError(f"skill paths use forward slashes; got {value!r}")
|
|
54
|
+
segments = value.split("/")
|
|
55
|
+
if any(segment in ("", ".", "..") for segment in segments):
|
|
56
|
+
raise ValueError(f"skill path is not a normalized relative path: {value!r}")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class SkillFile(ProtocolModel):
|
|
60
|
+
"""One file inside a skill, addressed by content."""
|
|
61
|
+
|
|
62
|
+
path: NonEmptyString
|
|
63
|
+
media_type: NonEmptyString
|
|
64
|
+
size: int = Field(ge=0)
|
|
65
|
+
digest: Digest
|
|
66
|
+
|
|
67
|
+
@model_validator(mode="after")
|
|
68
|
+
def _check_path(self) -> Self:
|
|
69
|
+
"""Reject anything that is not a normalized relative POSIX path."""
|
|
70
|
+
_check_relative_posix_path(self.path)
|
|
71
|
+
return self
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class SkillArtifact(ProtocolModel):
|
|
75
|
+
"""A complete candidate skill, addressed by content."""
|
|
76
|
+
|
|
77
|
+
schema_version: Literal["techtree.skill.v1alpha1"]
|
|
78
|
+
name: NonEmptyString
|
|
79
|
+
root_digest: Digest
|
|
80
|
+
archive_digest: Digest
|
|
81
|
+
files: list[SkillFile]
|
|
82
|
+
source_kind: Literal["manual"]
|
|
83
|
+
parent_skill_digest: Digest | None
|
|
84
|
+
|
|
85
|
+
@model_validator(mode="after")
|
|
86
|
+
def _check_file_list(self) -> Self:
|
|
87
|
+
"""Require the entry file, a stable order, and no repeated path."""
|
|
88
|
+
paths = [file.path for file in self.files]
|
|
89
|
+
if SKILL_ENTRY_FILE not in paths:
|
|
90
|
+
raise ValueError(f"a skill must contain {SKILL_ENTRY_FILE}")
|
|
91
|
+
if len(set(paths)) != len(paths):
|
|
92
|
+
raise ValueError("a skill lists each path exactly once")
|
|
93
|
+
if paths != sorted(paths):
|
|
94
|
+
raise ValueError(
|
|
95
|
+
"skill files are sorted by path so that the same tree always "
|
|
96
|
+
"produces the same digest"
|
|
97
|
+
)
|
|
98
|
+
return self
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
class PolicyAcceptanceRequirement(ProtocolModel):
|
|
102
|
+
"""The rights policy a participant is being asked to accept.
|
|
103
|
+
|
|
104
|
+
Decisions document 0003 A5. This states what must be accepted; it does not
|
|
105
|
+
record that anyone accepted it.
|
|
106
|
+
"""
|
|
107
|
+
|
|
108
|
+
data_policy_digest: Digest
|
|
109
|
+
required: bool
|
|
110
|
+
summary: NonEmptyString
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
class SubmissionDraft(ProtocolModel):
|
|
114
|
+
"""Everything a participant is about to commit to, in one object."""
|
|
115
|
+
|
|
116
|
+
schema_version: Literal["techtree.submission-draft.v1alpha1"]
|
|
117
|
+
id: NonEmptyString
|
|
118
|
+
campaign_spec_digest: Digest
|
|
119
|
+
program_ref: ProgramRef | None
|
|
120
|
+
public_context: PublicContext | None
|
|
121
|
+
data_policy_digest: Digest
|
|
122
|
+
outcome_contract_digest: Digest | None
|
|
123
|
+
skill_artifact: SkillArtifact
|
|
124
|
+
baseline_manifest_digest: Digest
|
|
125
|
+
candidate_manifest_digest: Digest
|
|
126
|
+
included_files: list[NonEmptyString]
|
|
127
|
+
estimated_episodes: int = Field(ge=1)
|
|
128
|
+
policy_acceptance: PolicyAcceptanceRequirement
|
|
129
|
+
warnings: list[NonEmptyString]
|
|
130
|
+
created_at: UtcDateTime
|
|
131
|
+
|
|
132
|
+
@model_validator(mode="after")
|
|
133
|
+
def _check_draft(self) -> Self:
|
|
134
|
+
"""Keep the draft self-consistent about rights, files, and variants."""
|
|
135
|
+
if self.policy_acceptance.data_policy_digest != self.data_policy_digest:
|
|
136
|
+
raise ValueError(
|
|
137
|
+
"the draft asks for acceptance of a different DataPolicy than "
|
|
138
|
+
"the one it runs under"
|
|
139
|
+
)
|
|
140
|
+
if self.baseline_manifest_digest == self.candidate_manifest_digest:
|
|
141
|
+
raise ValueError(
|
|
142
|
+
"baseline and candidate manifests must differ; an identical "
|
|
143
|
+
"pair measures nothing"
|
|
144
|
+
)
|
|
145
|
+
for path in self.included_files:
|
|
146
|
+
_check_relative_posix_path(path)
|
|
147
|
+
if self.included_files != sorted(self.included_files):
|
|
148
|
+
raise ValueError("included_files is sorted")
|
|
149
|
+
if len(set(self.included_files)) != len(self.included_files):
|
|
150
|
+
raise ValueError("included_files lists each path exactly once")
|
|
151
|
+
artifact_paths = [file.path for file in self.skill_artifact.files]
|
|
152
|
+
if self.included_files != artifact_paths:
|
|
153
|
+
raise ValueError(
|
|
154
|
+
"included_files must be exactly the files in the skill artifact"
|
|
155
|
+
)
|
|
156
|
+
return self
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
"""The result of one comparison. Spec section 11.11.
|
|
2
|
+
|
|
3
|
+
An ``UpliftReport`` answers one question: did the candidate satisfy the
|
|
4
|
+
scientific comparison contract? It does not answer whether anyone should deploy
|
|
5
|
+
the result. That is a release decision, it depends on cost, risk, and product
|
|
6
|
+
context that this object knows nothing about, and it belongs to a future
|
|
7
|
+
``ReleaseDecision``.
|
|
8
|
+
|
|
9
|
+
The five statuses are separate on purpose. A run can complete while its scores
|
|
10
|
+
are development-only; scores can be valid while evidence is partial; a
|
|
11
|
+
comparison can be controlled while publication is blocked by the data policy.
|
|
12
|
+
Collapsing them into one "status" would force a caller to guess which of those
|
|
13
|
+
five things a single word referred to.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from enum import StrEnum
|
|
19
|
+
from typing import Literal, Self
|
|
20
|
+
|
|
21
|
+
from pydantic import Field, model_validator
|
|
22
|
+
|
|
23
|
+
from techtree.models.base import Digest, NonEmptyString, ProtocolModel, UtcDateTime
|
|
24
|
+
from techtree.models.campaign import ProgramRef, PublicContext
|
|
25
|
+
from techtree.models.episode_receipt import EvidenceStatus, ScoreStatus
|
|
26
|
+
from techtree.models.evaluation_backend import EvaluationBackendSpec
|
|
27
|
+
from techtree.models.experiment import ManifestComparison
|
|
28
|
+
|
|
29
|
+
__all__ = [
|
|
30
|
+
"ComparisonStatus",
|
|
31
|
+
"ExecutionStatus",
|
|
32
|
+
"PrimaryUpliftResult",
|
|
33
|
+
"PublicationStatus",
|
|
34
|
+
"TaskDelta",
|
|
35
|
+
"UpliftDecision",
|
|
36
|
+
"UpliftReport",
|
|
37
|
+
"UpliftStatuses",
|
|
38
|
+
]
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class ExecutionStatus(StrEnum):
|
|
42
|
+
"""How the run itself ended."""
|
|
43
|
+
|
|
44
|
+
COMPLETED = "completed"
|
|
45
|
+
PARTIAL = "partial"
|
|
46
|
+
FAILED = "failed"
|
|
47
|
+
CANCELLED = "cancelled"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class ComparisonStatus(StrEnum):
|
|
51
|
+
"""Whether the two manifests differed only where permitted."""
|
|
52
|
+
|
|
53
|
+
PENDING = "pending"
|
|
54
|
+
CONTROLLED = "controlled"
|
|
55
|
+
CONTROLLED_WITH_WARNINGS = "controlled_with_warnings"
|
|
56
|
+
INVALID = "invalid"
|
|
57
|
+
DEVELOPMENT_ONLY = "development_only"
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class PublicationStatus(StrEnum):
|
|
61
|
+
"""Where the report stands with respect to being published."""
|
|
62
|
+
|
|
63
|
+
NOT_REQUESTED = "not_requested"
|
|
64
|
+
BLOCKED = "blocked"
|
|
65
|
+
PENDING = "pending"
|
|
66
|
+
PUBLISHED = "published"
|
|
67
|
+
FAILED = "failed"
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class UpliftDecision(StrEnum):
|
|
71
|
+
"""The verdict on the scientific comparison contract."""
|
|
72
|
+
|
|
73
|
+
ACCEPTED = "accepted"
|
|
74
|
+
REJECTED = "rejected"
|
|
75
|
+
INCONCLUSIVE = "inconclusive"
|
|
76
|
+
INVALID = "invalid"
|
|
77
|
+
DEVELOPMENT_ONLY = "development_only"
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class TaskDelta(ProtocolModel):
|
|
81
|
+
"""What one task contributed to the comparison."""
|
|
82
|
+
|
|
83
|
+
task_hash: Digest
|
|
84
|
+
baseline_reward: float
|
|
85
|
+
candidate_reward: float
|
|
86
|
+
delta: float
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
class PrimaryUpliftResult(ProtocolModel):
|
|
90
|
+
"""The headline comparison on the primary reward."""
|
|
91
|
+
|
|
92
|
+
reward_name: NonEmptyString
|
|
93
|
+
baseline_mean: float
|
|
94
|
+
candidate_mean: float
|
|
95
|
+
absolute_delta: float
|
|
96
|
+
relative_delta: float | None
|
|
97
|
+
wins: int = Field(ge=0)
|
|
98
|
+
losses: int = Field(ge=0)
|
|
99
|
+
ties: int = Field(ge=0)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
class UpliftStatuses(ProtocolModel):
|
|
103
|
+
"""The five independent statuses of a report."""
|
|
104
|
+
|
|
105
|
+
execution: ExecutionStatus
|
|
106
|
+
score: ScoreStatus
|
|
107
|
+
evidence: EvidenceStatus
|
|
108
|
+
comparison: ComparisonStatus
|
|
109
|
+
publication: PublicationStatus
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
class UpliftReport(ProtocolModel):
|
|
113
|
+
"""The complete result of one baseline-versus-candidate comparison."""
|
|
114
|
+
|
|
115
|
+
schema_version: Literal["techtree.uplift-report.v1alpha1"]
|
|
116
|
+
id: NonEmptyString
|
|
117
|
+
run_id: NonEmptyString
|
|
118
|
+
campaign_spec_digest: Digest
|
|
119
|
+
program_ref: ProgramRef | None
|
|
120
|
+
public_context: PublicContext | None
|
|
121
|
+
data_policy_digest: Digest
|
|
122
|
+
outcome_contract_digest: Digest | None
|
|
123
|
+
evaluation_backend: EvaluationBackendSpec
|
|
124
|
+
taskset_validation_receipt_digest: Digest
|
|
125
|
+
baseline_manifest_digest: Digest
|
|
126
|
+
candidate_manifest_digest: Digest
|
|
127
|
+
statuses: UpliftStatuses
|
|
128
|
+
manifest_comparison: ManifestComparison
|
|
129
|
+
primary_result: PrimaryUpliftResult
|
|
130
|
+
task_deltas: list[TaskDelta]
|
|
131
|
+
decision: UpliftDecision
|
|
132
|
+
proof_grade: Literal["development_only", "P1"]
|
|
133
|
+
publication_eligible: bool
|
|
134
|
+
created_at: UtcDateTime
|
|
135
|
+
|
|
136
|
+
@model_validator(mode="after")
|
|
137
|
+
def _check_report_cannot_overclaim(self) -> Self:
|
|
138
|
+
"""Keep a development-only report from presenting itself as evidence."""
|
|
139
|
+
development_only = self.proof_grade == "development_only"
|
|
140
|
+
if development_only and self.decision is not UpliftDecision.DEVELOPMENT_ONLY:
|
|
141
|
+
raise ValueError(
|
|
142
|
+
"a development_only report reaches a development_only decision"
|
|
143
|
+
)
|
|
144
|
+
if development_only and self.publication_eligible:
|
|
145
|
+
raise ValueError("a development_only report is never publication eligible")
|
|
146
|
+
if self.publication_eligible and self.statuses.publication is (
|
|
147
|
+
PublicationStatus.BLOCKED
|
|
148
|
+
):
|
|
149
|
+
raise ValueError(
|
|
150
|
+
"a report cannot be publication eligible while its publication "
|
|
151
|
+
"status is blocked"
|
|
152
|
+
)
|
|
153
|
+
if self.baseline_manifest_digest == self.candidate_manifest_digest:
|
|
154
|
+
raise ValueError(
|
|
155
|
+
"a report compares two different manifests; an identical pair "
|
|
156
|
+
"measures nothing"
|
|
157
|
+
)
|
|
158
|
+
return self
|