techtree 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- techtree/__init__.py +35 -0
- techtree/__main__.py +14 -0
- techtree/canonical.py +239 -0
- techtree/catalog/__init__.py +25 -0
- techtree/catalog/repository.py +400 -0
- techtree/catalog/service.py +419 -0
- techtree/cli/__init__.py +1 -0
- techtree/cli/app.py +416 -0
- techtree/cli/commands/__init__.py +1 -0
- techtree/cli/commands/climb.py +1223 -0
- techtree/cli/commands/doctor.py +147 -0
- techtree/cli/commands/engine.py +207 -0
- techtree/cli/commands/proof.py +556 -0
- techtree/cli/commands/publish.py +447 -0
- techtree/cli/commands/release.py +303 -0
- techtree/cli/commands/run.py +1067 -0
- techtree/cli/commands/setup.py +181 -0
- techtree/cli/commands/skill.py +221 -0
- techtree/cli/commands/uplift.py +698 -0
- techtree/cli/commands/withdraw.py +212 -0
- techtree/cli/confirm.py +47 -0
- techtree/cli/context.py +96 -0
- techtree/cli/invoke.py +220 -0
- techtree/cli/output.py +280 -0
- techtree/constants.py +138 -0
- techtree/crypto.py +128 -0
- techtree/doctor/__init__.py +1 -0
- techtree/doctor/checks.py +675 -0
- techtree/doctor/execution_checks.py +435 -0
- techtree/doctor/service.py +326 -0
- techtree/drafts/__init__.py +32 -0
- techtree/drafts/source.py +146 -0
- techtree/drafts/store.py +992 -0
- techtree/engines/__init__.py +1 -0
- techtree/engines/bundle.py +251 -0
- techtree/engines/installer.py +679 -0
- techtree/engines/registry.py +235 -0
- techtree/engines/runner.py +170 -0
- techtree/errors.py +262 -0
- techtree/fs.py +234 -0
- techtree/harness.py +108 -0
- techtree/identity/__init__.py +41 -0
- techtree/identity/models.py +113 -0
- techtree/identity/service.py +199 -0
- techtree/identity/store.py +263 -0
- techtree/ids.py +85 -0
- techtree/manifests/__init__.py +39 -0
- techtree/manifests/builder.py +433 -0
- techtree/manifests/compare.py +376 -0
- techtree/models/__init__.py +282 -0
- techtree/models/base.py +201 -0
- techtree/models/campaign.py +484 -0
- techtree/models/catalog.py +227 -0
- techtree/models/cli.py +151 -0
- techtree/models/climb.py +254 -0
- techtree/models/data_policy.py +130 -0
- techtree/models/engine.py +156 -0
- techtree/models/episode_receipt.py +130 -0
- techtree/models/evaluation_backend.py +113 -0
- techtree/models/experiment.py +154 -0
- techtree/models/run.py +214 -0
- techtree/models/skill.py +156 -0
- techtree/models/uplift_report.py +158 -0
- techtree/models/validation.py +299 -0
- techtree/paths.py +116 -0
- techtree/presentation/__init__.py +31 -0
- techtree/presentation/build.py +1242 -0
- techtree/presentation/compact.py +246 -0
- techtree/presentation/evidence.py +169 -0
- techtree/presentation/models.py +358 -0
- techtree/presentation/rich.py +312 -0
- techtree/presentation/sanitize.py +156 -0
- techtree/publication/__init__.py +44 -0
- techtree/publication/address.py +180 -0
- techtree/publication/coordinates.py +26 -0
- techtree/publication/journal.py +212 -0
- techtree/publication/keccak.py +183 -0
- techtree/publication/models.py +209 -0
- techtree/publication/offer.py +35 -0
- techtree/publication/service.py +618 -0
- techtree/publication/transport.py +296 -0
- techtree/publication/verify.py +242 -0
- techtree/publication/withdraw.py +156 -0
- techtree/py.typed +0 -0
- techtree/receipts/__init__.py +52 -0
- techtree/receipts/bundle.py +578 -0
- techtree/receipts/compare.py +1065 -0
- techtree/receipts/episode.py +672 -0
- techtree/receipts/execution.py +630 -0
- techtree/receipts/observed.py +474 -0
- techtree/receipts/set.py +336 -0
- techtree/receipts/uplift.py +655 -0
- techtree/receipts/verify.py +1055 -0
- techtree/release/__init__.py +9 -0
- techtree/release/bootstrap.py +509 -0
- techtree/release/checks.py +376 -0
- techtree/release/document.py +125 -0
- techtree/release/generate.py +221 -0
- techtree/release/models.py +293 -0
- techtree/release/provenance.py +109 -0
- techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
- techtree/resources/catalog/catalog.json +32 -0
- techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
- techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
- techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
- techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
- techtree/resources/engines/default/engine.json +20 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
- techtree/resources/engines/default/pyproject.toml +23 -0
- techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
- techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
- techtree/resources/engines/default/tools/normalize_validation.py +222 -0
- techtree/resources/engines/default/uv.lock +1758 -0
- techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
- techtree/resources/release/build-provenance.json +4 -0
- techtree/resources/release/release-core.json +24 -0
- techtree/runs/__init__.py +31 -0
- techtree/runs/artifacts.py +750 -0
- techtree/runs/child_registry.py +228 -0
- techtree/runs/events.py +478 -0
- techtree/runs/executor.py +140 -0
- techtree/runs/fake.py +741 -0
- techtree/runs/launcher.py +253 -0
- techtree/runs/machine.py +489 -0
- techtree/runs/real.py +789 -0
- techtree/runs/service.py +616 -0
- techtree/runs/store.py +555 -0
- techtree/runs/validation.py +259 -0
- techtree/runs/variants.py +684 -0
- techtree/settings.py +143 -0
- techtree/skills/__init__.py +14 -0
- techtree/skills/archive.py +282 -0
- techtree/skills/policy.py +62 -0
- techtree/skills/scanner.py +394 -0
- techtree/skills/service.py +752 -0
- techtree/skills/starter.py +434 -0
- techtree/tasksets/__init__.py +1 -0
- techtree/tasksets/membership.py +269 -0
- techtree/tasksets/provider.py +207 -0
- techtree/tasksets/resolver.py +311 -0
- techtree/tasksets/service.py +484 -0
- techtree/tasksets/verifiers_cli.py +538 -0
- techtree/uplift/__init__.py +20 -0
- techtree/uplift/context.py +544 -0
- techtree/uplift/derive.py +203 -0
- techtree/uplift/public_tasks.py +151 -0
- techtree/uplift/service.py +719 -0
- techtree/uplift/source.py +160 -0
- techtree/verifiers/__init__.py +31 -0
- techtree/verifiers/budget.py +219 -0
- techtree/verifiers/child.py +633 -0
- techtree/verifiers/compiler.py +432 -0
- techtree/verifiers/config.py +365 -0
- techtree/verifiers/credentials.py +321 -0
- techtree/verifiers/image.py +126 -0
- techtree/verifiers/models.py +527 -0
- techtree/verifiers/outputs.py +368 -0
- techtree/verifiers/progress.py +192 -0
- techtree/verifiers/supervisor.py +341 -0
- techtree/verifiers/verify.py +782 -0
- techtree/version.py +39 -0
- techtree/worker/__init__.py +18 -0
- techtree/worker/execute.py +487 -0
- techtree/worker/main.py +57 -0
- techtree-0.1.0.dist-info/METADATA +344 -0
- techtree-0.1.0.dist-info/RECORD +174 -0
- techtree-0.1.0.dist-info/WHEEL +4 -0
- techtree-0.1.0.dist-info/entry_points.txt +3 -0
- techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,358 @@
|
|
|
1
|
+
"""The channel-neutral shape of a result. Spec section 7.13.
|
|
2
|
+
|
|
3
|
+
Both renderers in this build — the terminal one and the compact one a gateway
|
|
4
|
+
relays — consume exactly this payload, and nothing else draws a result. The
|
|
5
|
+
payload is a contract about *what a result is*, never an assumption about how
|
|
6
|
+
anyone draws it: numbers, labels, outcomes, caveats and next steps, with no
|
|
7
|
+
markup, no colour and no channel anywhere in it.
|
|
8
|
+
|
|
9
|
+
Three properties are load-bearing.
|
|
10
|
+
|
|
11
|
+
*It is derived, never authored.* Every score, status and digest here is copied
|
|
12
|
+
out of a signed :class:`~techtree.models.uplift_report.UpliftReport`. Nothing
|
|
13
|
+
downstream can alter one, because nothing downstream is given the chance to
|
|
14
|
+
compute one.
|
|
15
|
+
|
|
16
|
+
*It is frozen.* The models are :class:`~techtree.models.base.ProtocolModel`
|
|
17
|
+
subclasses even though the payload is not part of the frozen v0.1 protocol,
|
|
18
|
+
which is why the schema version says ``presentation`` rather than a protocol
|
|
19
|
+
object's name. Freezing means two renderings of one report cannot disagree
|
|
20
|
+
because something mutated the payload between them.
|
|
21
|
+
|
|
22
|
+
*It carries nothing hidden.* A hidden expected answer, a grader's source, or a
|
|
23
|
+
credential has no field to enter through, and
|
|
24
|
+
:func:`~techtree.presentation.sanitize.ensure_no_hidden_task_material` checks
|
|
25
|
+
the free text that could carry one anyway.
|
|
26
|
+
|
|
27
|
+
One thing that is not a field lives here too: :class:`TaskDisplay`, the
|
|
28
|
+
reader's answer to how much of the per-task table they want, and
|
|
29
|
+
:func:`selected_task_rows`, which turns that answer into rows. It is here
|
|
30
|
+
rather than in either renderer because a reader who asks the same question of
|
|
31
|
+
two channels has to be shown the same tasks, and two renderers that each kept
|
|
32
|
+
their own idea of which rows "changed" means could quietly disagree about
|
|
33
|
+
whether a tie is a change.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from __future__ import annotations
|
|
37
|
+
|
|
38
|
+
from enum import StrEnum
|
|
39
|
+
from typing import Final, Literal, Self
|
|
40
|
+
|
|
41
|
+
from pydantic import Field, model_validator
|
|
42
|
+
|
|
43
|
+
from techtree.models.base import Digest, NonEmptyString, ProtocolModel
|
|
44
|
+
from techtree.models.cli import NextAction
|
|
45
|
+
from techtree.receipts.execution import CostProvenance
|
|
46
|
+
|
|
47
|
+
__all__ = [
|
|
48
|
+
"PRESENTATION_SCHEMA_VERSION",
|
|
49
|
+
"DerivedCost",
|
|
50
|
+
"EconomicsSource",
|
|
51
|
+
"PresentationCaveat",
|
|
52
|
+
"ScoreBar",
|
|
53
|
+
"SkillSummary",
|
|
54
|
+
"TaskDisplay",
|
|
55
|
+
"TaskOutcome",
|
|
56
|
+
"TaskResultRow",
|
|
57
|
+
"UpliftPresentationPayload",
|
|
58
|
+
"selected_task_rows",
|
|
59
|
+
]
|
|
60
|
+
|
|
61
|
+
#: Deliberately not in :mod:`techtree.constants`, which holds protocol schema
|
|
62
|
+
#: versions. A presentation payload is a view, and spec section 3.5 keeps views
|
|
63
|
+
#: out of the protocol.
|
|
64
|
+
PRESENTATION_SCHEMA_VERSION: Final = "techtree.presentation.uplift.v1"
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
type TaskOutcome = Literal["win", "loss", "tie"]
|
|
68
|
+
"""Which way one task moved between the two variants."""
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
type EconomicsSource = Literal[
|
|
72
|
+
"comparison_execution_record",
|
|
73
|
+
"episode_receipts",
|
|
74
|
+
"unavailable",
|
|
75
|
+
]
|
|
76
|
+
"""Where the cost and timing on a payload came from, if anywhere.
|
|
77
|
+
|
|
78
|
+
Decisions document 0007 R6 puts the comparison's economics in a signed record
|
|
79
|
+
of its own. A payload built from a run that has one says so; one built from a
|
|
80
|
+
run that does not says that instead, and never quietly presents a number whose
|
|
81
|
+
source it cannot name."""
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
class ScoreBar(ProtocolModel):
|
|
85
|
+
"""One score, and the text a renderer draws it as.
|
|
86
|
+
|
|
87
|
+
``display`` is computed once, in the builder, so that the terminal and a
|
|
88
|
+
phone message draw the same bar from the same string rather than each
|
|
89
|
+
inventing a scale.
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
label: NonEmptyString
|
|
93
|
+
value: float
|
|
94
|
+
maximum: float = Field(gt=0.0)
|
|
95
|
+
display: NonEmptyString
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class TaskResultRow(ProtocolModel):
|
|
99
|
+
"""What one committed task contributed to the comparison."""
|
|
100
|
+
|
|
101
|
+
position: int = Field(ge=0)
|
|
102
|
+
task_label: NonEmptyString
|
|
103
|
+
baseline_score: float
|
|
104
|
+
candidate_score: float
|
|
105
|
+
delta: float
|
|
106
|
+
outcome: TaskOutcome
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
class TaskDisplay(StrEnum):
|
|
110
|
+
"""Which task rows a reader asked to see."""
|
|
111
|
+
|
|
112
|
+
ALL = "all"
|
|
113
|
+
CHANGED = "changed"
|
|
114
|
+
REGRESSIONS = "regressions"
|
|
115
|
+
NONE = "none"
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
#: Losses first, then wins, then ties. Within a group, committed task order.
|
|
119
|
+
_OUTCOME_RANK: Final[dict[str, int]] = {"loss": 0, "win": 1, "tie": 2}
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def selected_task_rows(
|
|
123
|
+
rows: list[TaskResultRow], show: TaskDisplay
|
|
124
|
+
) -> list[TaskResultRow]:
|
|
125
|
+
"""Return the rows a reader asked for, worst first.
|
|
126
|
+
|
|
127
|
+
A reader scanning a table wants the rows that moved the wrong way, then the
|
|
128
|
+
ones that moved, then the rest, so the order is fixed here rather than left
|
|
129
|
+
to whichever channel is drawing. ``TaskDisplay.NONE`` selects nothing at
|
|
130
|
+
all, which is the honest reading of a reader who said they did not want the
|
|
131
|
+
table: a channel given no rows prints no table and no heading over it.
|
|
132
|
+
|
|
133
|
+
Selecting rows is all this does. No filter can change a count, because
|
|
134
|
+
every count a reader sees comes from the payload rather than from what a
|
|
135
|
+
channel happened to have room for.
|
|
136
|
+
"""
|
|
137
|
+
if show is TaskDisplay.NONE:
|
|
138
|
+
return []
|
|
139
|
+
if show is TaskDisplay.REGRESSIONS:
|
|
140
|
+
chosen = [row for row in rows if row.outcome == "loss"]
|
|
141
|
+
elif show is TaskDisplay.CHANGED:
|
|
142
|
+
chosen = [row for row in rows if row.outcome != "tie"]
|
|
143
|
+
else:
|
|
144
|
+
chosen = list(rows)
|
|
145
|
+
return sorted(chosen, key=lambda row: (_OUTCOME_RANK[row.outcome], row.position))
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
class SkillSummary(ProtocolModel):
|
|
149
|
+
"""One side's Skill, described by size and content address.
|
|
150
|
+
|
|
151
|
+
A baseline with no Skill is a real state rather than a missing value: it is
|
|
152
|
+
what a Skill-insertion comparison measures against, so it has a label and
|
|
153
|
+
no digest.
|
|
154
|
+
"""
|
|
155
|
+
|
|
156
|
+
label: NonEmptyString
|
|
157
|
+
root_digest: Digest | None
|
|
158
|
+
file_count: int = Field(ge=0)
|
|
159
|
+
total_bytes: int = Field(ge=0)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
class DerivedCost(ProtocolModel):
|
|
163
|
+
"""A dollar figure worked out while rendering, from what the run recorded.
|
|
164
|
+
|
|
165
|
+
Decisions document 0007 R6 forbids exactly one thing about cost: a figure
|
|
166
|
+
presented as better sourced than it is. This is not a bill and is never
|
|
167
|
+
drawn as one, so everything it rests on travels with it — the two token
|
|
168
|
+
counts that were multiplied, the prices they were multiplied by, and the
|
|
169
|
+
day those prices were read.
|
|
170
|
+
|
|
171
|
+
``cached_input_tokens`` and ``prices_name_a_cached_rate`` are carried
|
|
172
|
+
together because a provider that serves part of the prompt from its own
|
|
173
|
+
cache usually charges less for it. When the recorded prices name no cached
|
|
174
|
+
rate, every token is priced at the full rate and the reader is told the
|
|
175
|
+
figure is on the high side, which is the only direction an unstated
|
|
176
|
+
discount can move it. The count is ``None`` when the run recorded no
|
|
177
|
+
usable cache split, which is not the same as a run that cached nothing.
|
|
178
|
+
"""
|
|
179
|
+
|
|
180
|
+
usd: float = Field(ge=0.0)
|
|
181
|
+
input_tokens: int = Field(ge=0)
|
|
182
|
+
output_tokens: int = Field(ge=0)
|
|
183
|
+
cached_input_tokens: int | None = Field(default=None, ge=0)
|
|
184
|
+
prices_name_a_cached_rate: bool
|
|
185
|
+
model_id: NonEmptyString
|
|
186
|
+
input_usd_per_mtok: float = Field(gt=0.0)
|
|
187
|
+
output_usd_per_mtok: float = Field(gt=0.0)
|
|
188
|
+
prices_recorded_on: NonEmptyString
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
class PresentationCaveat(ProtocolModel):
|
|
192
|
+
"""One thing a reader must know before believing what they just read.
|
|
193
|
+
|
|
194
|
+
Caveats are part of the payload rather than of a renderer, so that a
|
|
195
|
+
channel cannot drop one by being short of room.
|
|
196
|
+
"""
|
|
197
|
+
|
|
198
|
+
code: NonEmptyString
|
|
199
|
+
severity: Literal["info", "warning", "error"]
|
|
200
|
+
text: NonEmptyString
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
class UpliftPresentationPayload(ProtocolModel):
|
|
204
|
+
"""One comparison, ready to be shown anywhere.
|
|
205
|
+
|
|
206
|
+
``comparison_label`` names which result in the chain this is;
|
|
207
|
+
``change_label`` names the one thing that differed between the two sides,
|
|
208
|
+
in the arrow form decisions document 0019 section 1 fixes. They are two
|
|
209
|
+
fields because they answer two questions — which receipt am I holding, and
|
|
210
|
+
what did it measure — and a channel with room for only one should not have
|
|
211
|
+
to guess which.
|
|
212
|
+
"""
|
|
213
|
+
|
|
214
|
+
schema_version: Literal["techtree.presentation.uplift.v1"]
|
|
215
|
+
run_id: NonEmptyString
|
|
216
|
+
campaign_title: NonEmptyString
|
|
217
|
+
comparison_label: NonEmptyString
|
|
218
|
+
change_label: NonEmptyString
|
|
219
|
+
baseline_skill: SkillSummary
|
|
220
|
+
candidate_skill: SkillSummary
|
|
221
|
+
baseline_score: float
|
|
222
|
+
candidate_score: float
|
|
223
|
+
absolute_delta: float
|
|
224
|
+
relative_delta: float | None
|
|
225
|
+
wins: int = Field(ge=0)
|
|
226
|
+
losses: int = Field(ge=0)
|
|
227
|
+
ties: int = Field(ge=0)
|
|
228
|
+
task_rows: list[TaskResultRow]
|
|
229
|
+
baseline_tasks_scored_full: int | None
|
|
230
|
+
candidate_tasks_scored_full: int | None
|
|
231
|
+
baseline_tokens: int | None
|
|
232
|
+
candidate_tokens: int | None
|
|
233
|
+
baseline_seconds: float | None
|
|
234
|
+
candidate_seconds: float | None
|
|
235
|
+
baseline_model_turns: int | None
|
|
236
|
+
candidate_model_turns: int | None
|
|
237
|
+
baseline_rate_limited_calls: int | None
|
|
238
|
+
candidate_rate_limited_calls: int | None
|
|
239
|
+
every_rollout_completed: bool | None
|
|
240
|
+
economics_source: EconomicsSource
|
|
241
|
+
cost_usd: float | None = Field(default=None, ge=0.0)
|
|
242
|
+
cost_provenance: CostProvenance
|
|
243
|
+
derived_cost: DerivedCost | None = None
|
|
244
|
+
cost_unavailable_reason: NonEmptyString | None = None
|
|
245
|
+
decision: NonEmptyString
|
|
246
|
+
proof_grade: NonEmptyString
|
|
247
|
+
verification_status: NonEmptyString
|
|
248
|
+
caveats: list[PresentationCaveat]
|
|
249
|
+
next_actions: list[NextAction]
|
|
250
|
+
|
|
251
|
+
@model_validator(mode="after")
|
|
252
|
+
def _check_the_rows_and_the_counts_describe_one_comparison(self) -> Self:
|
|
253
|
+
"""Reject a payload whose table and headline disagree."""
|
|
254
|
+
outcomes = [row.outcome for row in self.task_rows]
|
|
255
|
+
counts: tuple[tuple[TaskOutcome, int], ...] = (
|
|
256
|
+
("win", self.wins),
|
|
257
|
+
("loss", self.losses),
|
|
258
|
+
("tie", self.ties),
|
|
259
|
+
)
|
|
260
|
+
for outcome, count in counts:
|
|
261
|
+
if outcomes.count(outcome) != count:
|
|
262
|
+
raise ValueError(
|
|
263
|
+
f"the payload reports {count} {outcome} rows and carries "
|
|
264
|
+
f"{outcomes.count(outcome)}"
|
|
265
|
+
)
|
|
266
|
+
positions = [row.position for row in self.task_rows]
|
|
267
|
+
if positions != sorted(positions) or len(set(positions)) != len(positions):
|
|
268
|
+
raise ValueError(
|
|
269
|
+
"task rows are carried in committed task order, each position once"
|
|
270
|
+
)
|
|
271
|
+
return self
|
|
272
|
+
|
|
273
|
+
@model_validator(mode="after")
|
|
274
|
+
def _check_the_cost_is_never_shown_without_its_source(self) -> Self:
|
|
275
|
+
"""Reject a payload whose cost claims a provenance it does not have.
|
|
276
|
+
|
|
277
|
+
Decisions document 0007 R6 forbids exactly one thing about cost: a
|
|
278
|
+
figure presented as better sourced than it is. The shape enforces the
|
|
279
|
+
pair here so that no renderer has to remember to.
|
|
280
|
+
"""
|
|
281
|
+
known = self.cost_usd is not None
|
|
282
|
+
claims_source = self.cost_provenance is not CostProvenance.UNAVAILABLE
|
|
283
|
+
if known != claims_source:
|
|
284
|
+
raise ValueError(
|
|
285
|
+
"a cost figure needs a provenance and a provenance needs a "
|
|
286
|
+
f"figure; got {self.cost_usd!r} as {self.cost_provenance.value}"
|
|
287
|
+
)
|
|
288
|
+
if known and self.economics_source != "comparison_execution_record":
|
|
289
|
+
raise ValueError(
|
|
290
|
+
"a cost figure comes from the signed execution record; a "
|
|
291
|
+
f"payload sourced from {self.economics_source} has none"
|
|
292
|
+
)
|
|
293
|
+
return self
|
|
294
|
+
|
|
295
|
+
@model_validator(mode="after")
|
|
296
|
+
def _check_a_derived_cost_never_stands_beside_a_reported_one(self) -> Self:
|
|
297
|
+
"""Reject a payload carrying two answers to "what did this cost?".
|
|
298
|
+
|
|
299
|
+
A figure the provider reported is the better answer wherever there is
|
|
300
|
+
one, so a derived figure exists only in its absence. Two of them in one
|
|
301
|
+
payload would leave each channel free to pick, and two channels showing
|
|
302
|
+
one run would then be able to disagree about money.
|
|
303
|
+
"""
|
|
304
|
+
if self.derived_cost is not None and self.cost_usd is not None:
|
|
305
|
+
raise ValueError(
|
|
306
|
+
"a cost is derived only when none was reported; this payload "
|
|
307
|
+
f"carries both {self.derived_cost.usd} and {self.cost_usd}"
|
|
308
|
+
)
|
|
309
|
+
figure = self.derived_cost is not None or self.cost_usd is not None
|
|
310
|
+
if figure == (self.cost_unavailable_reason is not None):
|
|
311
|
+
raise ValueError(
|
|
312
|
+
"a payload with no cost figure says what is missing, and one "
|
|
313
|
+
"with a figure has nothing to explain away"
|
|
314
|
+
)
|
|
315
|
+
return self
|
|
316
|
+
|
|
317
|
+
@model_validator(mode="after")
|
|
318
|
+
def _check_the_counts_read_from_the_run_arrive_together(self) -> Self:
|
|
319
|
+
"""Reject a payload that read half of one run's recorded traces.
|
|
320
|
+
|
|
321
|
+
Turns, throttling and whether every rollout finished are one reading of
|
|
322
|
+
one pair of recorded, digest-checked files. A payload holding some of
|
|
323
|
+
them and not the others would be describing a reading that never
|
|
324
|
+
happened.
|
|
325
|
+
"""
|
|
326
|
+
read = (
|
|
327
|
+
self.baseline_model_turns,
|
|
328
|
+
self.candidate_model_turns,
|
|
329
|
+
self.baseline_rate_limited_calls,
|
|
330
|
+
self.candidate_rate_limited_calls,
|
|
331
|
+
self.every_rollout_completed,
|
|
332
|
+
)
|
|
333
|
+
if None in read and any(value is not None for value in read):
|
|
334
|
+
raise ValueError(
|
|
335
|
+
"the counts read from a run's recorded traces are all present "
|
|
336
|
+
f"or all absent; got {read}"
|
|
337
|
+
)
|
|
338
|
+
return self
|
|
339
|
+
|
|
340
|
+
@model_validator(mode="after")
|
|
341
|
+
def _check_the_task_counts_fit_the_table(self) -> Self:
|
|
342
|
+
"""Reject a headline count that no per-task table could produce."""
|
|
343
|
+
baseline = self.baseline_tasks_scored_full
|
|
344
|
+
candidate = self.candidate_tasks_scored_full
|
|
345
|
+
if baseline is None or candidate is None:
|
|
346
|
+
if baseline is not candidate:
|
|
347
|
+
raise ValueError(
|
|
348
|
+
"both sides carry a task count or neither does; got "
|
|
349
|
+
f"{(baseline, candidate)}"
|
|
350
|
+
)
|
|
351
|
+
return self
|
|
352
|
+
for count in (baseline, candidate):
|
|
353
|
+
if not 0 <= count <= len(self.task_rows):
|
|
354
|
+
raise ValueError(
|
|
355
|
+
f"a side scored between 0 and {len(self.task_rows)} of the "
|
|
356
|
+
f"comparison's tasks; got {count}"
|
|
357
|
+
)
|
|
358
|
+
return self
|
|
@@ -0,0 +1,312 @@
|
|
|
1
|
+
"""The terminal rendering. Spec section 7.15.
|
|
2
|
+
|
|
3
|
+
This is the CLI's own renderer, and in the released product it is the whole
|
|
4
|
+
terminal result path: a signed report becomes a neutral payload, and this draws
|
|
5
|
+
it. No model is asked to explain a result, so nothing between the numbers and
|
|
6
|
+
the reader can change one by drawing it.
|
|
7
|
+
|
|
8
|
+
Four rules, and each one is about a reader rather than about a terminal.
|
|
9
|
+
|
|
10
|
+
*Meaning is never carried by colour alone.* Every outcome is spelled ``WIN``,
|
|
11
|
+
``LOSS`` or ``TIE``, every caveat is introduced by its severity in words, and a
|
|
12
|
+
terminal with no colour at all loses nothing but decoration. ``NO_COLOR`` and
|
|
13
|
+
``--no-color`` are honoured by the console the CLI builds, and Rich emits no
|
|
14
|
+
escape sequences at all when stdout is not a terminal.
|
|
15
|
+
|
|
16
|
+
*The rendering is deterministic.* No spinner, no progress animation, no clock,
|
|
17
|
+
no dictionary iteration order: the same payload renders to the same bytes,
|
|
18
|
+
which is what makes it testable at all.
|
|
19
|
+
|
|
20
|
+
*The order is the order of trust.* What was compared, then the result, then the
|
|
21
|
+
tasks, then what changed, then the caveats — so a reader who stops early stops
|
|
22
|
+
having read the honest version. The caveats are never last-resort small print
|
|
23
|
+
they can miss; a development-only or failed-verification result says so in the
|
|
24
|
+
badge at the top as well.
|
|
25
|
+
|
|
26
|
+
*Regressions come first.* A reader scanning a table wants the rows that moved
|
|
27
|
+
the wrong way, then the ones that moved, then the rest.
|
|
28
|
+
|
|
29
|
+
The payload's next actions are deliberately not drawn here. Every Techtree
|
|
30
|
+
command ends with the same numbered next-steps block, rendered by the CLI from
|
|
31
|
+
the envelope it returns, and a result that grew a second one of its own would
|
|
32
|
+
be the only command in the product that answers that question twice.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
from __future__ import annotations
|
|
36
|
+
|
|
37
|
+
from typing import Final
|
|
38
|
+
|
|
39
|
+
from rich.console import Console
|
|
40
|
+
from rich.table import Table
|
|
41
|
+
|
|
42
|
+
from techtree.presentation.build import (
|
|
43
|
+
HELD_FIXED_LINE,
|
|
44
|
+
NOT_BROAD_CAPABILITY_LINE,
|
|
45
|
+
VERIFICATION_FAILED,
|
|
46
|
+
VERIFICATION_NOT_VERIFIED,
|
|
47
|
+
VERIFICATION_VERIFIED,
|
|
48
|
+
cost_explanation,
|
|
49
|
+
cost_summary,
|
|
50
|
+
decision_headline,
|
|
51
|
+
efficiency_sentence,
|
|
52
|
+
score_bars,
|
|
53
|
+
solved_line,
|
|
54
|
+
task_count_line,
|
|
55
|
+
)
|
|
56
|
+
from techtree.presentation.models import (
|
|
57
|
+
TaskDisplay,
|
|
58
|
+
UpliftPresentationPayload,
|
|
59
|
+
selected_task_rows,
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
__all__ = ["outcome_label", "render_uplift_console"]
|
|
63
|
+
|
|
64
|
+
_SEVERITY_PREFIX: Final[dict[str, str]] = {
|
|
65
|
+
"info": "Note:",
|
|
66
|
+
"warning": "Warning:",
|
|
67
|
+
"error": "Error:",
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
_SEVERITY_STYLE: Final[dict[str, str]] = {
|
|
71
|
+
"info": "",
|
|
72
|
+
"warning": "yellow",
|
|
73
|
+
"error": "red",
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
_OUTCOME_LABEL: Final[dict[str, str]] = {
|
|
77
|
+
"win": "WIN",
|
|
78
|
+
"loss": "LOSS",
|
|
79
|
+
"tie": "TIE",
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
_VERIFICATION_BADGE: Final[dict[str, str]] = {
|
|
83
|
+
VERIFICATION_VERIFIED: "local proof verified offline",
|
|
84
|
+
VERIFICATION_FAILED: "LOCAL PROOF DID NOT VERIFY",
|
|
85
|
+
VERIFICATION_NOT_VERIFIED: "local proof not checked",
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def outcome_label(outcome: str) -> str:
|
|
90
|
+
"""Return the word a reader sees for one task's outcome."""
|
|
91
|
+
return _OUTCOME_LABEL[outcome]
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def render_uplift_console(
|
|
95
|
+
payload: UpliftPresentationPayload,
|
|
96
|
+
console: Console,
|
|
97
|
+
*,
|
|
98
|
+
show_tasks: TaskDisplay = TaskDisplay.CHANGED,
|
|
99
|
+
) -> None:
|
|
100
|
+
"""Render an accessible, side-by-side terminal result.
|
|
101
|
+
|
|
102
|
+
``show_tasks`` is the reader's choice of how much of the per-task table to
|
|
103
|
+
print (spec section 7.21). It selects rows and nothing else: no filter here
|
|
104
|
+
can change a count, because the counts come from the payload.
|
|
105
|
+
"""
|
|
106
|
+
_header(payload, console)
|
|
107
|
+
_primary(payload, console)
|
|
108
|
+
_tasks(payload, console, show_tasks)
|
|
109
|
+
_efficiency(payload, console)
|
|
110
|
+
_what_changed(payload, console)
|
|
111
|
+
_caveats(payload, console)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _header(payload: UpliftPresentationPayload, console: Console) -> None:
|
|
115
|
+
"""Campaign, comparison, and the badge that says how much this is worth."""
|
|
116
|
+
console.print(payload.campaign_title)
|
|
117
|
+
console.print(payload.comparison_label)
|
|
118
|
+
console.print(
|
|
119
|
+
f"[{payload.proof_grade} · {_VERIFICATION_BADGE[payload.verification_status]}]"
|
|
120
|
+
)
|
|
121
|
+
console.print()
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _primary(payload: UpliftPresentationPayload, console: Console) -> None:
|
|
125
|
+
"""What was established, what is still failing, and the numbers under it.
|
|
126
|
+
|
|
127
|
+
The three lines at the top are the ones a reader repeats to somebody else,
|
|
128
|
+
so they are the three that have to be defensible on their own: what this
|
|
129
|
+
comparison showed, how much of the task family is still unsolved, and the
|
|
130
|
+
fact that none of it is evidence about broad capability.
|
|
131
|
+
|
|
132
|
+
Wins, losses and ties do not appear here. On this Climb the baseline scores
|
|
133
|
+
nothing on every task, so a tie means the candidate failed the task too,
|
|
134
|
+
and a headline of twenty-four wins and no losses would read as a clean
|
|
135
|
+
sweep over a run with twelve tasks still failing. The three words stay in
|
|
136
|
+
the per-task table below, where each one is beside the scores that produced
|
|
137
|
+
it.
|
|
138
|
+
|
|
139
|
+
The count leads the detail because it is the number a person reads a result
|
|
140
|
+
in. A mean of 0.667 and "24 of 36" are the same measurement, and only one
|
|
141
|
+
of them can be repeated to somebody over a table.
|
|
142
|
+
"""
|
|
143
|
+
console.print(f"Result {decision_headline(payload)}")
|
|
144
|
+
console.print(f" {solved_line(payload)}")
|
|
145
|
+
console.print(f" {NOT_BROAD_CAPABILITY_LINE}")
|
|
146
|
+
console.print()
|
|
147
|
+
|
|
148
|
+
counted = task_count_line(payload)
|
|
149
|
+
if counted is not None:
|
|
150
|
+
console.print(f"Tasks {counted}")
|
|
151
|
+
console.print()
|
|
152
|
+
|
|
153
|
+
table = Table(box=None, show_header=False, pad_edge=False, padding=(0, 2))
|
|
154
|
+
table.add_column("side", no_wrap=True)
|
|
155
|
+
table.add_column("bar", no_wrap=True)
|
|
156
|
+
for bar in score_bars(payload):
|
|
157
|
+
table.add_row(bar.label, bar.display)
|
|
158
|
+
console.print(table)
|
|
159
|
+
console.print()
|
|
160
|
+
|
|
161
|
+
console.print(
|
|
162
|
+
f"Change {payload.absolute_delta:+.3f} mean score ({_relative(payload)})"
|
|
163
|
+
)
|
|
164
|
+
console.print()
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _relative(payload: UpliftPresentationPayload) -> str:
|
|
168
|
+
"""Say what a relative change is, or why there is not one."""
|
|
169
|
+
if payload.relative_delta is None:
|
|
170
|
+
return "no relative change: the baseline scored nothing"
|
|
171
|
+
return f"{payload.relative_delta:+.1%} relative"
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _tasks(
|
|
175
|
+
payload: UpliftPresentationPayload, console: Console, show: TaskDisplay
|
|
176
|
+
) -> None:
|
|
177
|
+
"""The per-task table, regressions first."""
|
|
178
|
+
if show is TaskDisplay.NONE:
|
|
179
|
+
return
|
|
180
|
+
rows = selected_task_rows(payload.task_rows, show)
|
|
181
|
+
if not rows:
|
|
182
|
+
console.print(_nothing_selected(show))
|
|
183
|
+
console.print()
|
|
184
|
+
return
|
|
185
|
+
|
|
186
|
+
table = Table(box=None, pad_edge=False, padding=(0, 2))
|
|
187
|
+
table.add_column("Task", no_wrap=True)
|
|
188
|
+
table.add_column("Baseline", justify="right", no_wrap=True)
|
|
189
|
+
table.add_column("Candidate", justify="right", no_wrap=True)
|
|
190
|
+
table.add_column("Change", justify="right", no_wrap=True)
|
|
191
|
+
table.add_column("Outcome", no_wrap=True)
|
|
192
|
+
for row in rows:
|
|
193
|
+
table.add_row(
|
|
194
|
+
row.task_label,
|
|
195
|
+
f"{row.baseline_score:.3f}",
|
|
196
|
+
f"{row.candidate_score:.3f}",
|
|
197
|
+
f"{row.delta:+.3f}",
|
|
198
|
+
outcome_label(row.outcome),
|
|
199
|
+
)
|
|
200
|
+
console.print(table)
|
|
201
|
+
if len(rows) != len(payload.task_rows):
|
|
202
|
+
console.print(
|
|
203
|
+
f"Showing {len(rows)} of {len(payload.task_rows)} tasks. "
|
|
204
|
+
"Use --show-tasks all for the rest."
|
|
205
|
+
)
|
|
206
|
+
console.print()
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _nothing_selected(show: TaskDisplay) -> str:
|
|
210
|
+
"""Say why the table is empty, in terms of what was asked for.
|
|
211
|
+
|
|
212
|
+
An empty selection and an unchanged comparison are not the same fact, and
|
|
213
|
+
one sentence cannot honestly stand for both. Asking a run that solved
|
|
214
|
+
twenty-three tasks to list its regressions finds none, and answering that
|
|
215
|
+
with "every task scored the same" contradicts the three lines directly
|
|
216
|
+
above it — the reader is told the run improved and then told it did not.
|
|
217
|
+
So each selection says what its own emptiness means.
|
|
218
|
+
"""
|
|
219
|
+
if show is TaskDisplay.REGRESSIONS:
|
|
220
|
+
return "No task scored worse with the Skill than without it."
|
|
221
|
+
if show is TaskDisplay.ALL:
|
|
222
|
+
return "This run recorded no tasks."
|
|
223
|
+
return "Every task scored the same on both sides."
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _efficiency(payload: UpliftPresentationPayload, console: Console) -> None:
|
|
227
|
+
"""Turns, tokens, time and cost, each shown with the source it came from.
|
|
228
|
+
|
|
229
|
+
Decisions document 0007 R6 governs the cost line: a figure is never printed
|
|
230
|
+
without saying where it is from, because "$4.10" and "$4.10, estimated" are
|
|
231
|
+
different claims and only one of them is about money that was actually
|
|
232
|
+
charged.
|
|
233
|
+
|
|
234
|
+
The sentence under the numbers is there because two durations side by side
|
|
235
|
+
are not a finding. What the two sides did differently is legible in how
|
|
236
|
+
many times each had to go back to the model, and that is the half of the
|
|
237
|
+
comparison a different machine would reproduce.
|
|
238
|
+
"""
|
|
239
|
+
seconds = _pair(payload.baseline_seconds, payload.candidate_seconds, unit="s")
|
|
240
|
+
console.print("Efficiency")
|
|
241
|
+
turns = _pair(payload.baseline_model_turns, payload.candidate_model_turns)
|
|
242
|
+
console.print(f" Turns {turns}")
|
|
243
|
+
console.print(
|
|
244
|
+
f" Tokens {_pair(payload.baseline_tokens, payload.candidate_tokens)}"
|
|
245
|
+
)
|
|
246
|
+
console.print(f" Time {seconds}")
|
|
247
|
+
console.print(f" Cost {cost_summary(payload)}")
|
|
248
|
+
for line in cost_explanation(payload):
|
|
249
|
+
console.print(f" {line}")
|
|
250
|
+
console.print(f" Source {_ECONOMICS_SOURCE[payload.economics_source]}")
|
|
251
|
+
sentence = efficiency_sentence(payload)
|
|
252
|
+
if sentence is not None:
|
|
253
|
+
console.print(f" {sentence}")
|
|
254
|
+
console.print()
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def _pair(
|
|
258
|
+
baseline: float | int | None, candidate: float | int | None, *, unit: str = ""
|
|
259
|
+
) -> str:
|
|
260
|
+
"""Return both sides of one measurement, or say it was not recorded."""
|
|
261
|
+
if baseline is None and candidate is None:
|
|
262
|
+
return "not recorded for this run"
|
|
263
|
+
return f"baseline {_number(baseline, unit)}, candidate {_number(candidate, unit)}"
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _number(value: float | int | None, unit: str) -> str:
|
|
267
|
+
"""Return one measurement, or the word for an absent one.
|
|
268
|
+
|
|
269
|
+
Whole counts are grouped. A token total is seven digits in a real run, and
|
|
270
|
+
seven ungrouped digits are a number a reader has to count rather than read.
|
|
271
|
+
"""
|
|
272
|
+
if value is None:
|
|
273
|
+
return "unavailable"
|
|
274
|
+
if isinstance(value, float):
|
|
275
|
+
return f"{value:.1f}{unit}"
|
|
276
|
+
return f"{value:,}{unit}"
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
def _what_changed(payload: UpliftPresentationPayload, console: Console) -> None:
|
|
280
|
+
"""The one declared difference, and the statement that it is the only one."""
|
|
281
|
+
console.print("What changed")
|
|
282
|
+
console.print(f" {payload.change_label}")
|
|
283
|
+
for side, skill in (
|
|
284
|
+
("Baseline ", payload.baseline_skill),
|
|
285
|
+
("Candidate", payload.candidate_skill),
|
|
286
|
+
):
|
|
287
|
+
digest = "none" if skill.root_digest is None else skill.root_digest
|
|
288
|
+
console.print(f" {side} {skill.label}")
|
|
289
|
+
console.print(f" {digest}")
|
|
290
|
+
console.print(f" {HELD_FIXED_LINE}")
|
|
291
|
+
console.print(" Each of those was checked against what the run actually did.")
|
|
292
|
+
console.print()
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
#: Where the numbers above came from, said in the reader's terms.
|
|
296
|
+
_ECONOMICS_SOURCE: Final[dict[str, str]] = {
|
|
297
|
+
"comparison_execution_record": "this run's signed execution record",
|
|
298
|
+
"episode_receipts": "the run's receipts; no execution record was written",
|
|
299
|
+
"unavailable": "nothing recorded it",
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def _caveats(payload: UpliftPresentationPayload, console: Console) -> None:
|
|
304
|
+
"""Every caveat, in payload order, introduced by severity in words."""
|
|
305
|
+
if not payload.caveats:
|
|
306
|
+
return
|
|
307
|
+
console.print("What this does and does not prove")
|
|
308
|
+
for caveat in payload.caveats:
|
|
309
|
+
console.print(
|
|
310
|
+
f" {_SEVERITY_PREFIX[caveat.severity]} {caveat.text}",
|
|
311
|
+
style=_SEVERITY_STYLE[caveat.severity] or None,
|
|
312
|
+
)
|