techtree 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- techtree/__init__.py +35 -0
- techtree/__main__.py +14 -0
- techtree/canonical.py +239 -0
- techtree/catalog/__init__.py +25 -0
- techtree/catalog/repository.py +400 -0
- techtree/catalog/service.py +419 -0
- techtree/cli/__init__.py +1 -0
- techtree/cli/app.py +416 -0
- techtree/cli/commands/__init__.py +1 -0
- techtree/cli/commands/climb.py +1223 -0
- techtree/cli/commands/doctor.py +147 -0
- techtree/cli/commands/engine.py +207 -0
- techtree/cli/commands/proof.py +556 -0
- techtree/cli/commands/publish.py +447 -0
- techtree/cli/commands/release.py +303 -0
- techtree/cli/commands/run.py +1067 -0
- techtree/cli/commands/setup.py +181 -0
- techtree/cli/commands/skill.py +221 -0
- techtree/cli/commands/uplift.py +698 -0
- techtree/cli/commands/withdraw.py +212 -0
- techtree/cli/confirm.py +47 -0
- techtree/cli/context.py +96 -0
- techtree/cli/invoke.py +220 -0
- techtree/cli/output.py +280 -0
- techtree/constants.py +138 -0
- techtree/crypto.py +128 -0
- techtree/doctor/__init__.py +1 -0
- techtree/doctor/checks.py +675 -0
- techtree/doctor/execution_checks.py +435 -0
- techtree/doctor/service.py +326 -0
- techtree/drafts/__init__.py +32 -0
- techtree/drafts/source.py +146 -0
- techtree/drafts/store.py +992 -0
- techtree/engines/__init__.py +1 -0
- techtree/engines/bundle.py +251 -0
- techtree/engines/installer.py +679 -0
- techtree/engines/registry.py +235 -0
- techtree/engines/runner.py +170 -0
- techtree/errors.py +262 -0
- techtree/fs.py +234 -0
- techtree/harness.py +108 -0
- techtree/identity/__init__.py +41 -0
- techtree/identity/models.py +113 -0
- techtree/identity/service.py +199 -0
- techtree/identity/store.py +263 -0
- techtree/ids.py +85 -0
- techtree/manifests/__init__.py +39 -0
- techtree/manifests/builder.py +433 -0
- techtree/manifests/compare.py +376 -0
- techtree/models/__init__.py +282 -0
- techtree/models/base.py +201 -0
- techtree/models/campaign.py +484 -0
- techtree/models/catalog.py +227 -0
- techtree/models/cli.py +151 -0
- techtree/models/climb.py +254 -0
- techtree/models/data_policy.py +130 -0
- techtree/models/engine.py +156 -0
- techtree/models/episode_receipt.py +130 -0
- techtree/models/evaluation_backend.py +113 -0
- techtree/models/experiment.py +154 -0
- techtree/models/run.py +214 -0
- techtree/models/skill.py +156 -0
- techtree/models/uplift_report.py +158 -0
- techtree/models/validation.py +299 -0
- techtree/paths.py +116 -0
- techtree/presentation/__init__.py +31 -0
- techtree/presentation/build.py +1242 -0
- techtree/presentation/compact.py +246 -0
- techtree/presentation/evidence.py +169 -0
- techtree/presentation/models.py +358 -0
- techtree/presentation/rich.py +312 -0
- techtree/presentation/sanitize.py +156 -0
- techtree/publication/__init__.py +44 -0
- techtree/publication/address.py +180 -0
- techtree/publication/coordinates.py +26 -0
- techtree/publication/journal.py +212 -0
- techtree/publication/keccak.py +183 -0
- techtree/publication/models.py +209 -0
- techtree/publication/offer.py +35 -0
- techtree/publication/service.py +618 -0
- techtree/publication/transport.py +296 -0
- techtree/publication/verify.py +242 -0
- techtree/publication/withdraw.py +156 -0
- techtree/py.typed +0 -0
- techtree/receipts/__init__.py +52 -0
- techtree/receipts/bundle.py +578 -0
- techtree/receipts/compare.py +1065 -0
- techtree/receipts/episode.py +672 -0
- techtree/receipts/execution.py +630 -0
- techtree/receipts/observed.py +474 -0
- techtree/receipts/set.py +336 -0
- techtree/receipts/uplift.py +655 -0
- techtree/receipts/verify.py +1055 -0
- techtree/release/__init__.py +9 -0
- techtree/release/bootstrap.py +509 -0
- techtree/release/checks.py +376 -0
- techtree/release/document.py +125 -0
- techtree/release/generate.py +221 -0
- techtree/release/models.py +293 -0
- techtree/release/provenance.py +109 -0
- techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
- techtree/resources/catalog/catalog.json +32 -0
- techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
- techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
- techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
- techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
- techtree/resources/engines/default/engine.json +20 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
- techtree/resources/engines/default/pyproject.toml +23 -0
- techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
- techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
- techtree/resources/engines/default/tools/normalize_validation.py +222 -0
- techtree/resources/engines/default/uv.lock +1758 -0
- techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
- techtree/resources/release/build-provenance.json +4 -0
- techtree/resources/release/release-core.json +24 -0
- techtree/runs/__init__.py +31 -0
- techtree/runs/artifacts.py +750 -0
- techtree/runs/child_registry.py +228 -0
- techtree/runs/events.py +478 -0
- techtree/runs/executor.py +140 -0
- techtree/runs/fake.py +741 -0
- techtree/runs/launcher.py +253 -0
- techtree/runs/machine.py +489 -0
- techtree/runs/real.py +789 -0
- techtree/runs/service.py +616 -0
- techtree/runs/store.py +555 -0
- techtree/runs/validation.py +259 -0
- techtree/runs/variants.py +684 -0
- techtree/settings.py +143 -0
- techtree/skills/__init__.py +14 -0
- techtree/skills/archive.py +282 -0
- techtree/skills/policy.py +62 -0
- techtree/skills/scanner.py +394 -0
- techtree/skills/service.py +752 -0
- techtree/skills/starter.py +434 -0
- techtree/tasksets/__init__.py +1 -0
- techtree/tasksets/membership.py +269 -0
- techtree/tasksets/provider.py +207 -0
- techtree/tasksets/resolver.py +311 -0
- techtree/tasksets/service.py +484 -0
- techtree/tasksets/verifiers_cli.py +538 -0
- techtree/uplift/__init__.py +20 -0
- techtree/uplift/context.py +544 -0
- techtree/uplift/derive.py +203 -0
- techtree/uplift/public_tasks.py +151 -0
- techtree/uplift/service.py +719 -0
- techtree/uplift/source.py +160 -0
- techtree/verifiers/__init__.py +31 -0
- techtree/verifiers/budget.py +219 -0
- techtree/verifiers/child.py +633 -0
- techtree/verifiers/compiler.py +432 -0
- techtree/verifiers/config.py +365 -0
- techtree/verifiers/credentials.py +321 -0
- techtree/verifiers/image.py +126 -0
- techtree/verifiers/models.py +527 -0
- techtree/verifiers/outputs.py +368 -0
- techtree/verifiers/progress.py +192 -0
- techtree/verifiers/supervisor.py +341 -0
- techtree/verifiers/verify.py +782 -0
- techtree/version.py +39 -0
- techtree/worker/__init__.py +18 -0
- techtree/worker/execute.py +487 -0
- techtree/worker/main.py +57 -0
- techtree-0.1.0.dist-info/METADATA +344 -0
- techtree-0.1.0.dist-info/RECORD +174 -0
- techtree-0.1.0.dist-info/WHEEL +4 -0
- techtree-0.1.0.dist-info/entry_points.txt +3 -0
- techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
"""The run's own verified Skill text. Decisions document 0007 R2.
|
|
2
|
+
|
|
3
|
+
A host agent that is asked to revise a Skill has to read the Skill first, and
|
|
4
|
+
R2 says which copy it reads: the one the run owns, re-verified against the
|
|
5
|
+
run's own artifact at the moment it is read. This module is that read.
|
|
6
|
+
|
|
7
|
+
Three things are deliberate.
|
|
8
|
+
|
|
9
|
+
*Verification happens on the bytes that are returned.* The run's inputs are
|
|
10
|
+
checked file by file when they are loaded, but a check performed on one read
|
|
11
|
+
says nothing about a later one. The entrypoint is read once, hashed, and
|
|
12
|
+
compared with the artifact's own entry for it, and it is that same buffer that
|
|
13
|
+
becomes the returned text. There is no window between proving and using.
|
|
14
|
+
|
|
15
|
+
*The whole tree is proved, not just the file.* The artifact's file list is
|
|
16
|
+
re-digested against its root digest before any file is opened, and then every
|
|
17
|
+
file it lists is read and hashed — not only the entrypoint. A caller who is
|
|
18
|
+
handed text is being told two things at once: this is the entrypoint of a
|
|
19
|
+
Skill, and that Skill is the one the run measured. The second half would be
|
|
20
|
+
false if a reference file could be edited without anyone noticing, and a
|
|
21
|
+
Skill is a handful of small text files, so proving all of it costs nothing.
|
|
22
|
+
|
|
23
|
+
*A mismatch is a refusal, never a repair.* Nothing here rewrites a digest,
|
|
24
|
+
falls back to the archive, or returns text it could not vouch for. The only
|
|
25
|
+
outcomes are verified text and a typed error, because the consumer of this
|
|
26
|
+
text sends it to a model and binds a proposal to the digests beside it.
|
|
27
|
+
|
|
28
|
+
Composing the internal path to a run's staged Skill is what this exists to
|
|
29
|
+
prevent: the layout under a run directory is Techtree's, and a second
|
|
30
|
+
implementation of it somewhere else would be a second thing to keep true.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
from __future__ import annotations
|
|
34
|
+
|
|
35
|
+
from dataclasses import dataclass
|
|
36
|
+
from pathlib import Path
|
|
37
|
+
from typing import Final
|
|
38
|
+
|
|
39
|
+
from techtree.canonical import sha256_digest_bytes
|
|
40
|
+
from techtree.drafts.source import StagedSkill
|
|
41
|
+
from techtree.errors import VerificationError
|
|
42
|
+
from techtree.manifests.builder import skill_content_digest
|
|
43
|
+
from techtree.models.base import Digest
|
|
44
|
+
from techtree.models.skill import SKILL_ENTRY_FILE, SkillFile
|
|
45
|
+
|
|
46
|
+
__all__ = [
|
|
47
|
+
"SOURCE_SKILL_UNREADABLE",
|
|
48
|
+
"SOURCE_SKILL_UNVERIFIED",
|
|
49
|
+
"VerifiedSourceSkill",
|
|
50
|
+
"read_verified_source_skill",
|
|
51
|
+
]
|
|
52
|
+
|
|
53
|
+
#: The Skill this run owns does not hash to what the run says it is. Stable,
|
|
54
|
+
#: because a consumer branches on it: this is the one condition under which a
|
|
55
|
+
#: revision must not be proposed at all.
|
|
56
|
+
SOURCE_SKILL_UNVERIFIED: Final = "source_skill_unverified"
|
|
57
|
+
|
|
58
|
+
#: The bytes are there and correct but cannot be handed over as text — an
|
|
59
|
+
#: unreadable file, or an entrypoint that is not UTF-8. Different from a
|
|
60
|
+
#: mismatch, because nothing is claiming to be something it is not.
|
|
61
|
+
SOURCE_SKILL_UNREADABLE: Final = "source_skill_unreadable"
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@dataclass(frozen=True)
|
|
65
|
+
class VerifiedSourceSkill:
|
|
66
|
+
"""One run-owned Skill's entrypoint text, and what it was verified against."""
|
|
67
|
+
|
|
68
|
+
run_id: str
|
|
69
|
+
name: str
|
|
70
|
+
root_digest: Digest
|
|
71
|
+
entrypoint_path: str
|
|
72
|
+
entrypoint_digest: Digest
|
|
73
|
+
entrypoint_size: int
|
|
74
|
+
entrypoint_text: str
|
|
75
|
+
file_count: int
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def read_verified_source_skill(
|
|
79
|
+
staged: StagedSkill, *, run_id: str
|
|
80
|
+
) -> VerifiedSourceSkill:
|
|
81
|
+
"""Read one run-owned Skill's entrypoint, proving it as it is read."""
|
|
82
|
+
skill = staged.artifact
|
|
83
|
+
|
|
84
|
+
recomputed_root = skill_content_digest(skill.files)
|
|
85
|
+
_require(
|
|
86
|
+
recomputed_root == skill.root_digest,
|
|
87
|
+
"this run's copy of the Skill lists files that do not describe the "
|
|
88
|
+
"Skill it says it is",
|
|
89
|
+
run_id=run_id,
|
|
90
|
+
expected=skill.root_digest,
|
|
91
|
+
computed=recomputed_root,
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
entry: SkillFile | None = None
|
|
95
|
+
entrypoint_bytes = b""
|
|
96
|
+
for file in skill.files:
|
|
97
|
+
data = _read(staged.files / file.path, run_id=run_id, path=file.path)
|
|
98
|
+
computed = sha256_digest_bytes(data)
|
|
99
|
+
_require(
|
|
100
|
+
len(data) == file.size and computed == file.digest,
|
|
101
|
+
f"this run's copy of {file.path} is not the file the run measured",
|
|
102
|
+
run_id=run_id,
|
|
103
|
+
path=file.path,
|
|
104
|
+
expected=file.digest,
|
|
105
|
+
computed=computed,
|
|
106
|
+
)
|
|
107
|
+
if file.path == SKILL_ENTRY_FILE:
|
|
108
|
+
entry, entrypoint_bytes = file, data
|
|
109
|
+
|
|
110
|
+
if entry is None:
|
|
111
|
+
raise VerificationError(
|
|
112
|
+
f"this run's copy of the Skill lists no {SKILL_ENTRY_FILE}, so it "
|
|
113
|
+
"has no text to read",
|
|
114
|
+
code=SOURCE_SKILL_UNVERIFIED,
|
|
115
|
+
details={"run_id": run_id, "skill": skill.root_digest},
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
try:
|
|
119
|
+
text = entrypoint_bytes.decode("utf-8")
|
|
120
|
+
except UnicodeDecodeError as error:
|
|
121
|
+
raise VerificationError(
|
|
122
|
+
f"this run's copy of {entry.path} is not UTF-8 text, so it cannot "
|
|
123
|
+
"be read as a Skill",
|
|
124
|
+
code=SOURCE_SKILL_UNREADABLE,
|
|
125
|
+
details={"run_id": run_id, "path": entry.path},
|
|
126
|
+
) from error
|
|
127
|
+
|
|
128
|
+
return VerifiedSourceSkill(
|
|
129
|
+
run_id=run_id,
|
|
130
|
+
name=skill.name,
|
|
131
|
+
root_digest=skill.root_digest,
|
|
132
|
+
entrypoint_path=entry.path,
|
|
133
|
+
entrypoint_digest=entry.digest,
|
|
134
|
+
entrypoint_size=entry.size,
|
|
135
|
+
entrypoint_text=text,
|
|
136
|
+
file_count=len(skill.files),
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _read(location: Path, *, run_id: str, path: str) -> bytes:
|
|
141
|
+
"""Return one staged file's bytes, or refuse because it cannot be read."""
|
|
142
|
+
try:
|
|
143
|
+
return location.read_bytes()
|
|
144
|
+
except OSError as error:
|
|
145
|
+
raise VerificationError(
|
|
146
|
+
f"this run's copy of {path} could not be read: {error.strerror or error}",
|
|
147
|
+
code=SOURCE_SKILL_UNREADABLE,
|
|
148
|
+
details={"run_id": run_id, "path": path},
|
|
149
|
+
) from error
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _require(condition: bool, message: str, **details: str | int) -> None:
|
|
153
|
+
"""Refuse, with the two digests that disagree, unless they agree."""
|
|
154
|
+
if condition:
|
|
155
|
+
return
|
|
156
|
+
raise VerificationError(
|
|
157
|
+
message,
|
|
158
|
+
code=SOURCE_SKILL_UNVERIFIED,
|
|
159
|
+
details={key: str(value) for key, value in details.items()},
|
|
160
|
+
)
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""Native Verifiers execution. Spec sections 6.3-6.14.
|
|
2
|
+
|
|
3
|
+
This package is the boundary between a resolved Techtree experiment and the
|
|
4
|
+
pinned Verifiers ``eval`` entrypoint. Nothing here imports ``verifiers``: the
|
|
5
|
+
library belongs to the managed engine environment, and everything Techtree
|
|
6
|
+
knows about its wire shapes was proven empirically and written down in
|
|
7
|
+
``docs/verifiers-eval.md``. That document, not this code's optimism, is the
|
|
8
|
+
source for every assumption below.
|
|
9
|
+
|
|
10
|
+
The division of labour is deliberate:
|
|
11
|
+
|
|
12
|
+
``models``
|
|
13
|
+
Local integration types. Not protocol roots, not published.
|
|
14
|
+
``config``
|
|
15
|
+
The strict, allow-listed TOML Techtree is permitted to emit. The compiler
|
|
16
|
+
cannot express a Verifiers knob this module does not model, which is what
|
|
17
|
+
makes "only the skill differs" checkable rather than hoped for.
|
|
18
|
+
``compiler``
|
|
19
|
+
One resolved experiment in, one ``EvalToml`` out, deterministically.
|
|
20
|
+
``credentials``
|
|
21
|
+
Whether the declared endpoint can authenticate, answered without ever
|
|
22
|
+
returning, logging, or persisting a secret.
|
|
23
|
+
``outputs``
|
|
24
|
+
The files a finished run must have left behind, hashed exactly.
|
|
25
|
+
``verify``
|
|
26
|
+
Whether a compiled configuration survives the engine's own resolution.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
__all__: list[str] = []
|
|
@@ -0,0 +1,219 @@
|
|
|
1
|
+
"""Whether a Campaign's declared budgets are budgets. Decisions document 0029.
|
|
2
|
+
|
|
3
|
+
Two questions, asked before a run of an executable public Campaign starts and
|
|
4
|
+
never after.
|
|
5
|
+
|
|
6
|
+
*Is every declared limit enforceable?* A Campaign that names an output ceiling
|
|
7
|
+
and nothing else is not bounded in turns, in input, or in time, and the fields
|
|
8
|
+
it left empty read like caps to anybody looking at the document.
|
|
9
|
+
:func:`require_executable_budget` refuses that Campaign by name rather than
|
|
10
|
+
letting the run discover it as a bill. The three declared fields are the three
|
|
11
|
+
publisher decisions — turns, input, output; the total is derived from the last
|
|
12
|
+
two in the compiler and is not a fourth thing anyone chooses.
|
|
13
|
+
|
|
14
|
+
*Does the declared spending limit hold, given those limits?* The token exposure
|
|
15
|
+
of a comparison is a finite number the moment the limits are finite, so the
|
|
16
|
+
dollar exposure is that number times a price. :func:`calculate_release_cost_bound`
|
|
17
|
+
computes it deliberately high — the whole context window on top of the declared
|
|
18
|
+
input allowance, one more sampled reply on top of the declared output allowance
|
|
19
|
+
— because a bound that is only usually right is not a bound. If the result is
|
|
20
|
+
above what the Campaign says it may spend, the run does not start.
|
|
21
|
+
|
|
22
|
+
The prices are not protocol values. They are a fact about a provider's rate
|
|
23
|
+
card on a particular day, so they live in a small record written at release
|
|
24
|
+
time (``release/price-profile.json``) whose numbers this module carries and a
|
|
25
|
+
test compares. Decision 0029 says the honest form of the published claim out
|
|
26
|
+
loud: token exposure is protocol-bounded, and the dollar figure uses provider
|
|
27
|
+
pricing recorded at release time.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
import math
|
|
33
|
+
from typing import Final
|
|
34
|
+
|
|
35
|
+
from pydantic import Field
|
|
36
|
+
|
|
37
|
+
from techtree.errors import PrerequisiteError
|
|
38
|
+
from techtree.models.base import JsonValue, NonEmptyString, ProtocolModel
|
|
39
|
+
from techtree.models.campaign import CampaignSpec
|
|
40
|
+
|
|
41
|
+
__all__ = [
|
|
42
|
+
"CAMPAIGN_BUDGET_NOT_ENFORCED",
|
|
43
|
+
"CAMPAIGN_COST_BOUND_EXCEEDED",
|
|
44
|
+
"PRICE_PROFILE_SCHEMA_VERSION",
|
|
45
|
+
"RELEASE_PRICE_PROFILES",
|
|
46
|
+
"SUBJECT_PRICE_PROFILE_MISSING",
|
|
47
|
+
"PriceProfile",
|
|
48
|
+
"calculate_release_cost_bound",
|
|
49
|
+
"price_profile_for",
|
|
50
|
+
"require_cost_bound",
|
|
51
|
+
"require_executable_budget",
|
|
52
|
+
]
|
|
53
|
+
|
|
54
|
+
#: Stable error codes. Decisions document 0029, layer A.
|
|
55
|
+
CAMPAIGN_BUDGET_NOT_ENFORCED: Final = "campaign_budget_not_enforced"
|
|
56
|
+
CAMPAIGN_COST_BOUND_EXCEEDED: Final = "campaign_cost_bound_exceeded"
|
|
57
|
+
SUBJECT_PRICE_PROFILE_MISSING: Final = "subject_price_profile_missing"
|
|
58
|
+
|
|
59
|
+
PRICE_PROFILE_SCHEMA_VERSION: Final = "techtree.price-profile.v1"
|
|
60
|
+
|
|
61
|
+
#: How many sides one comparison executes. Both variants run the same taskset,
|
|
62
|
+
#: so the comparison's exposure is twice one variant's.
|
|
63
|
+
_VARIANTS_PER_COMPARISON: Final = 2
|
|
64
|
+
|
|
65
|
+
_TOKENS_PER_MILLION: Final = 1_000_000.0
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class PriceProfile(ProtocolModel):
|
|
69
|
+
"""What one subject model costs, and how much of it can be sent at once.
|
|
70
|
+
|
|
71
|
+
``context_window_tokens`` is not a price; it is the other half of the input
|
|
72
|
+
bound. Verifiers' ``max_input_tokens`` is checked between turns, so a single
|
|
73
|
+
turn may still carry a whole context window of prompt on top of whatever
|
|
74
|
+
the allowance had already accounted for. Charging for that overshoot is
|
|
75
|
+
what makes the bound a bound.
|
|
76
|
+
"""
|
|
77
|
+
|
|
78
|
+
schema_version: NonEmptyString
|
|
79
|
+
model_id: NonEmptyString
|
|
80
|
+
input_usd_per_mtok: float = Field(gt=0.0)
|
|
81
|
+
output_usd_per_mtok: float = Field(gt=0.0)
|
|
82
|
+
context_window_tokens: int = Field(ge=1)
|
|
83
|
+
source: NonEmptyString
|
|
84
|
+
recorded_on: NonEmptyString
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
#: The prices this release was bounded with, recorded on the day they were
|
|
88
|
+
#: read. The same numbers are committed as ``release/price-profile.json``, and
|
|
89
|
+
#: a test fails if the two ever disagree.
|
|
90
|
+
#:
|
|
91
|
+
#: ``context_window_tokens`` is deliberately conservative. The pinned provider
|
|
92
|
+
#: publishes no context ceiling for this model in anything this repository
|
|
93
|
+
#: carries, so the release uses 131072 — the largest window in common use for
|
|
94
|
+
#: models of this class — rather than a number nobody can point at. It only
|
|
95
|
+
#: ever makes the computed bound larger, so an unrecorded larger window would
|
|
96
|
+
#: be visible as a run that reached a limit, never as an underestimate that
|
|
97
|
+
#: passed unnoticed.
|
|
98
|
+
RELEASE_PRICE_PROFILES: Final[tuple[PriceProfile, ...]] = (
|
|
99
|
+
PriceProfile(
|
|
100
|
+
schema_version=PRICE_PROFILE_SCHEMA_VERSION,
|
|
101
|
+
model_id="qwen/qwen3.7-flash",
|
|
102
|
+
input_usd_per_mtok=0.03,
|
|
103
|
+
output_usd_per_mtok=0.13,
|
|
104
|
+
context_window_tokens=131072,
|
|
105
|
+
source="Prime Intellect published rate card for qwen/qwen3.7-flash",
|
|
106
|
+
recorded_on="2026-08-20",
|
|
107
|
+
),
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def price_profile_for(model_id: str) -> PriceProfile:
|
|
112
|
+
"""Return the recorded prices for one subject model, or refuse.
|
|
113
|
+
|
|
114
|
+
A missing profile is a refusal rather than a default. Guessing a price
|
|
115
|
+
would produce a dollar bound nobody could check, which is the one thing the
|
|
116
|
+
published claim may not be.
|
|
117
|
+
"""
|
|
118
|
+
for profile in RELEASE_PRICE_PROFILES:
|
|
119
|
+
if profile.model_id == model_id:
|
|
120
|
+
return profile
|
|
121
|
+
raise PrerequisiteError(
|
|
122
|
+
f"this release records no provider prices for {model_id}, so no "
|
|
123
|
+
"spending bound can be computed for it",
|
|
124
|
+
code=SUBJECT_PRICE_PROFILE_MISSING,
|
|
125
|
+
details={"model_id": model_id},
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def require_executable_budget(campaign: CampaignSpec) -> None:
|
|
130
|
+
"""Refuse a Campaign whose declared execution limits are not enforced.
|
|
131
|
+
|
|
132
|
+
The four values here are the four the engine can be made to enforce: the
|
|
133
|
+
rollout timeout, and the three token and turn allowances. A public Campaign
|
|
134
|
+
that leaves any of them empty is asking to be run without a bound on
|
|
135
|
+
something it appears to bound.
|
|
136
|
+
"""
|
|
137
|
+
missing: list[str] = []
|
|
138
|
+
if campaign.execution.timeout_seconds <= 0:
|
|
139
|
+
missing.append("execution.timeout_seconds")
|
|
140
|
+
if campaign.budgets.maximum_model_calls is None:
|
|
141
|
+
missing.append("budgets.maximum_model_calls")
|
|
142
|
+
if campaign.budgets.maximum_input_tokens is None:
|
|
143
|
+
missing.append("budgets.maximum_input_tokens")
|
|
144
|
+
if campaign.budgets.maximum_output_tokens is None:
|
|
145
|
+
missing.append("budgets.maximum_output_tokens")
|
|
146
|
+
if missing:
|
|
147
|
+
raise PrerequisiteError(
|
|
148
|
+
"this public Campaign has unenforced or missing execution limits",
|
|
149
|
+
code=CAMPAIGN_BUDGET_NOT_ENFORCED,
|
|
150
|
+
details={
|
|
151
|
+
"campaign_id": campaign.metadata.id,
|
|
152
|
+
"missing": list(missing),
|
|
153
|
+
},
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def calculate_release_cost_bound(
|
|
158
|
+
campaign: CampaignSpec, price_profile: PriceProfile
|
|
159
|
+
) -> float:
|
|
160
|
+
"""Return the most one comparison of this Campaign can cost, in dollars.
|
|
161
|
+
|
|
162
|
+
Per episode and per variant, with ``I`` and ``O`` the declared input and
|
|
163
|
+
output allowances, ``C`` the context window and ``S`` the per-call sampling
|
|
164
|
+
cap::
|
|
165
|
+
|
|
166
|
+
input ≤ I + C output ≤ O + S
|
|
167
|
+
bound = tasks × 2 × ((I + C)·Pᵢ + (O + S)·Pₒ)
|
|
168
|
+
|
|
169
|
+
The two overshoot terms are the one-turn soft overshoot. The pinned build
|
|
170
|
+
states the rule itself: the caps are checked between turns, so the turn
|
|
171
|
+
that crosses one still completes. One turn's worth is at most a whole
|
|
172
|
+
context window in and one sampled reply out. The rates are the highest
|
|
173
|
+
uncached ones the profile records.
|
|
174
|
+
"""
|
|
175
|
+
require_executable_budget(campaign)
|
|
176
|
+
budgets = campaign.budgets
|
|
177
|
+
assert budgets.maximum_input_tokens is not None
|
|
178
|
+
assert budgets.maximum_output_tokens is not None
|
|
179
|
+
|
|
180
|
+
episodes = campaign.taskset.selection.num_tasks * _VARIANTS_PER_COMPARISON
|
|
181
|
+
input_tokens = budgets.maximum_input_tokens + price_profile.context_window_tokens
|
|
182
|
+
output_tokens = budgets.maximum_output_tokens + campaign.subject.sampling.max_tokens
|
|
183
|
+
per_episode = (
|
|
184
|
+
input_tokens * price_profile.input_usd_per_mtok
|
|
185
|
+
+ output_tokens * price_profile.output_usd_per_mtok
|
|
186
|
+
) / _TOKENS_PER_MILLION
|
|
187
|
+
return episodes * per_episode
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def require_cost_bound(campaign: CampaignSpec, price_profile: PriceProfile) -> float:
|
|
191
|
+
"""Refuse a Campaign that can cost more than it says it may.
|
|
192
|
+
|
|
193
|
+
Returns the computed bound so a caller can record it. A Campaign that
|
|
194
|
+
declares no spending limit has no precondition to check here; what bounds
|
|
195
|
+
it is the token exposure above, and that is finite either way.
|
|
196
|
+
"""
|
|
197
|
+
bound = calculate_release_cost_bound(campaign, price_profile)
|
|
198
|
+
ceiling = campaign.budgets.maximum_usd
|
|
199
|
+
if ceiling is None or bound <= ceiling:
|
|
200
|
+
return bound
|
|
201
|
+
details: dict[str, JsonValue] = {
|
|
202
|
+
"campaign_id": campaign.metadata.id,
|
|
203
|
+
"calculated_bound_usd": _rounded(bound),
|
|
204
|
+
"maximum_usd": ceiling,
|
|
205
|
+
"model_id": price_profile.model_id,
|
|
206
|
+
"prices_recorded_on": price_profile.recorded_on,
|
|
207
|
+
}
|
|
208
|
+
raise PrerequisiteError(
|
|
209
|
+
"the most this comparison can cost under its own enforced limits — "
|
|
210
|
+
f"${_rounded(bound):.2f} at the prices this release recorded — is "
|
|
211
|
+
f"above the ${ceiling:.2f} the Campaign declares it may spend",
|
|
212
|
+
code=CAMPAIGN_COST_BOUND_EXCEEDED,
|
|
213
|
+
details=details,
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _rounded(amount: float) -> float:
|
|
218
|
+
"""Round a dollar figure up to the cent, so a bound is never reported low."""
|
|
219
|
+
return math.ceil(amount * 100.0) / 100.0
|