techtree 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. techtree/__init__.py +35 -0
  2. techtree/__main__.py +14 -0
  3. techtree/canonical.py +239 -0
  4. techtree/catalog/__init__.py +25 -0
  5. techtree/catalog/repository.py +400 -0
  6. techtree/catalog/service.py +419 -0
  7. techtree/cli/__init__.py +1 -0
  8. techtree/cli/app.py +416 -0
  9. techtree/cli/commands/__init__.py +1 -0
  10. techtree/cli/commands/climb.py +1223 -0
  11. techtree/cli/commands/doctor.py +147 -0
  12. techtree/cli/commands/engine.py +207 -0
  13. techtree/cli/commands/proof.py +556 -0
  14. techtree/cli/commands/publish.py +447 -0
  15. techtree/cli/commands/release.py +303 -0
  16. techtree/cli/commands/run.py +1067 -0
  17. techtree/cli/commands/setup.py +181 -0
  18. techtree/cli/commands/skill.py +221 -0
  19. techtree/cli/commands/uplift.py +698 -0
  20. techtree/cli/commands/withdraw.py +212 -0
  21. techtree/cli/confirm.py +47 -0
  22. techtree/cli/context.py +96 -0
  23. techtree/cli/invoke.py +220 -0
  24. techtree/cli/output.py +280 -0
  25. techtree/constants.py +138 -0
  26. techtree/crypto.py +128 -0
  27. techtree/doctor/__init__.py +1 -0
  28. techtree/doctor/checks.py +675 -0
  29. techtree/doctor/execution_checks.py +435 -0
  30. techtree/doctor/service.py +326 -0
  31. techtree/drafts/__init__.py +32 -0
  32. techtree/drafts/source.py +146 -0
  33. techtree/drafts/store.py +992 -0
  34. techtree/engines/__init__.py +1 -0
  35. techtree/engines/bundle.py +251 -0
  36. techtree/engines/installer.py +679 -0
  37. techtree/engines/registry.py +235 -0
  38. techtree/engines/runner.py +170 -0
  39. techtree/errors.py +262 -0
  40. techtree/fs.py +234 -0
  41. techtree/harness.py +108 -0
  42. techtree/identity/__init__.py +41 -0
  43. techtree/identity/models.py +113 -0
  44. techtree/identity/service.py +199 -0
  45. techtree/identity/store.py +263 -0
  46. techtree/ids.py +85 -0
  47. techtree/manifests/__init__.py +39 -0
  48. techtree/manifests/builder.py +433 -0
  49. techtree/manifests/compare.py +376 -0
  50. techtree/models/__init__.py +282 -0
  51. techtree/models/base.py +201 -0
  52. techtree/models/campaign.py +484 -0
  53. techtree/models/catalog.py +227 -0
  54. techtree/models/cli.py +151 -0
  55. techtree/models/climb.py +254 -0
  56. techtree/models/data_policy.py +130 -0
  57. techtree/models/engine.py +156 -0
  58. techtree/models/episode_receipt.py +130 -0
  59. techtree/models/evaluation_backend.py +113 -0
  60. techtree/models/experiment.py +154 -0
  61. techtree/models/run.py +214 -0
  62. techtree/models/skill.py +156 -0
  63. techtree/models/uplift_report.py +158 -0
  64. techtree/models/validation.py +299 -0
  65. techtree/paths.py +116 -0
  66. techtree/presentation/__init__.py +31 -0
  67. techtree/presentation/build.py +1242 -0
  68. techtree/presentation/compact.py +246 -0
  69. techtree/presentation/evidence.py +169 -0
  70. techtree/presentation/models.py +358 -0
  71. techtree/presentation/rich.py +312 -0
  72. techtree/presentation/sanitize.py +156 -0
  73. techtree/publication/__init__.py +44 -0
  74. techtree/publication/address.py +180 -0
  75. techtree/publication/coordinates.py +26 -0
  76. techtree/publication/journal.py +212 -0
  77. techtree/publication/keccak.py +183 -0
  78. techtree/publication/models.py +209 -0
  79. techtree/publication/offer.py +35 -0
  80. techtree/publication/service.py +618 -0
  81. techtree/publication/transport.py +296 -0
  82. techtree/publication/verify.py +242 -0
  83. techtree/publication/withdraw.py +156 -0
  84. techtree/py.typed +0 -0
  85. techtree/receipts/__init__.py +52 -0
  86. techtree/receipts/bundle.py +578 -0
  87. techtree/receipts/compare.py +1065 -0
  88. techtree/receipts/episode.py +672 -0
  89. techtree/receipts/execution.py +630 -0
  90. techtree/receipts/observed.py +474 -0
  91. techtree/receipts/set.py +336 -0
  92. techtree/receipts/uplift.py +655 -0
  93. techtree/receipts/verify.py +1055 -0
  94. techtree/release/__init__.py +9 -0
  95. techtree/release/bootstrap.py +509 -0
  96. techtree/release/checks.py +376 -0
  97. techtree/release/document.py +125 -0
  98. techtree/release/generate.py +221 -0
  99. techtree/release/models.py +293 -0
  100. techtree/release/provenance.py +109 -0
  101. techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
  102. techtree/resources/catalog/catalog.json +32 -0
  103. techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
  104. techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
  105. techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
  106. techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
  107. techtree/resources/engines/default/engine.json +20 -0
  108. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
  109. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
  110. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
  111. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
  112. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
  113. techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
  114. techtree/resources/engines/default/pyproject.toml +23 -0
  115. techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
  116. techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
  117. techtree/resources/engines/default/tools/normalize_validation.py +222 -0
  118. techtree/resources/engines/default/uv.lock +1758 -0
  119. techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
  120. techtree/resources/release/build-provenance.json +4 -0
  121. techtree/resources/release/release-core.json +24 -0
  122. techtree/runs/__init__.py +31 -0
  123. techtree/runs/artifacts.py +750 -0
  124. techtree/runs/child_registry.py +228 -0
  125. techtree/runs/events.py +478 -0
  126. techtree/runs/executor.py +140 -0
  127. techtree/runs/fake.py +741 -0
  128. techtree/runs/launcher.py +253 -0
  129. techtree/runs/machine.py +489 -0
  130. techtree/runs/real.py +789 -0
  131. techtree/runs/service.py +616 -0
  132. techtree/runs/store.py +555 -0
  133. techtree/runs/validation.py +259 -0
  134. techtree/runs/variants.py +684 -0
  135. techtree/settings.py +143 -0
  136. techtree/skills/__init__.py +14 -0
  137. techtree/skills/archive.py +282 -0
  138. techtree/skills/policy.py +62 -0
  139. techtree/skills/scanner.py +394 -0
  140. techtree/skills/service.py +752 -0
  141. techtree/skills/starter.py +434 -0
  142. techtree/tasksets/__init__.py +1 -0
  143. techtree/tasksets/membership.py +269 -0
  144. techtree/tasksets/provider.py +207 -0
  145. techtree/tasksets/resolver.py +311 -0
  146. techtree/tasksets/service.py +484 -0
  147. techtree/tasksets/verifiers_cli.py +538 -0
  148. techtree/uplift/__init__.py +20 -0
  149. techtree/uplift/context.py +544 -0
  150. techtree/uplift/derive.py +203 -0
  151. techtree/uplift/public_tasks.py +151 -0
  152. techtree/uplift/service.py +719 -0
  153. techtree/uplift/source.py +160 -0
  154. techtree/verifiers/__init__.py +31 -0
  155. techtree/verifiers/budget.py +219 -0
  156. techtree/verifiers/child.py +633 -0
  157. techtree/verifiers/compiler.py +432 -0
  158. techtree/verifiers/config.py +365 -0
  159. techtree/verifiers/credentials.py +321 -0
  160. techtree/verifiers/image.py +126 -0
  161. techtree/verifiers/models.py +527 -0
  162. techtree/verifiers/outputs.py +368 -0
  163. techtree/verifiers/progress.py +192 -0
  164. techtree/verifiers/supervisor.py +341 -0
  165. techtree/verifiers/verify.py +782 -0
  166. techtree/version.py +39 -0
  167. techtree/worker/__init__.py +18 -0
  168. techtree/worker/execute.py +487 -0
  169. techtree/worker/main.py +57 -0
  170. techtree-0.1.0.dist-info/METADATA +344 -0
  171. techtree-0.1.0.dist-info/RECORD +174 -0
  172. techtree-0.1.0.dist-info/WHEEL +4 -0
  173. techtree-0.1.0.dist-info/entry_points.txt +3 -0
  174. techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,358 @@
1
+ """The channel-neutral shape of a result. Spec section 7.13.
2
+
3
+ Both renderers in this build — the terminal one and the compact one a gateway
4
+ relays — consume exactly this payload, and nothing else draws a result. The
5
+ payload is a contract about *what a result is*, never an assumption about how
6
+ anyone draws it: numbers, labels, outcomes, caveats and next steps, with no
7
+ markup, no colour and no channel anywhere in it.
8
+
9
+ Three properties are load-bearing.
10
+
11
+ *It is derived, never authored.* Every score, status and digest here is copied
12
+ out of a signed :class:`~techtree.models.uplift_report.UpliftReport`. Nothing
13
+ downstream can alter one, because nothing downstream is given the chance to
14
+ compute one.
15
+
16
+ *It is frozen.* The models are :class:`~techtree.models.base.ProtocolModel`
17
+ subclasses even though the payload is not part of the frozen v0.1 protocol,
18
+ which is why the schema version says ``presentation`` rather than a protocol
19
+ object's name. Freezing means two renderings of one report cannot disagree
20
+ because something mutated the payload between them.
21
+
22
+ *It carries nothing hidden.* A hidden expected answer, a grader's source, or a
23
+ credential has no field to enter through, and
24
+ :func:`~techtree.presentation.sanitize.ensure_no_hidden_task_material` checks
25
+ the free text that could carry one anyway.
26
+
27
+ One thing that is not a field lives here too: :class:`TaskDisplay`, the
28
+ reader's answer to how much of the per-task table they want, and
29
+ :func:`selected_task_rows`, which turns that answer into rows. It is here
30
+ rather than in either renderer because a reader who asks the same question of
31
+ two channels has to be shown the same tasks, and two renderers that each kept
32
+ their own idea of which rows "changed" means could quietly disagree about
33
+ whether a tie is a change.
34
+ """
35
+
36
+ from __future__ import annotations
37
+
38
+ from enum import StrEnum
39
+ from typing import Final, Literal, Self
40
+
41
+ from pydantic import Field, model_validator
42
+
43
+ from techtree.models.base import Digest, NonEmptyString, ProtocolModel
44
+ from techtree.models.cli import NextAction
45
+ from techtree.receipts.execution import CostProvenance
46
+
47
+ __all__ = [
48
+ "PRESENTATION_SCHEMA_VERSION",
49
+ "DerivedCost",
50
+ "EconomicsSource",
51
+ "PresentationCaveat",
52
+ "ScoreBar",
53
+ "SkillSummary",
54
+ "TaskDisplay",
55
+ "TaskOutcome",
56
+ "TaskResultRow",
57
+ "UpliftPresentationPayload",
58
+ "selected_task_rows",
59
+ ]
60
+
61
+ #: Deliberately not in :mod:`techtree.constants`, which holds protocol schema
62
+ #: versions. A presentation payload is a view, and spec section 3.5 keeps views
63
+ #: out of the protocol.
64
+ PRESENTATION_SCHEMA_VERSION: Final = "techtree.presentation.uplift.v1"
65
+
66
+
67
+ type TaskOutcome = Literal["win", "loss", "tie"]
68
+ """Which way one task moved between the two variants."""
69
+
70
+
71
+ type EconomicsSource = Literal[
72
+ "comparison_execution_record",
73
+ "episode_receipts",
74
+ "unavailable",
75
+ ]
76
+ """Where the cost and timing on a payload came from, if anywhere.
77
+
78
+ Decisions document 0007 R6 puts the comparison's economics in a signed record
79
+ of its own. A payload built from a run that has one says so; one built from a
80
+ run that does not says that instead, and never quietly presents a number whose
81
+ source it cannot name."""
82
+
83
+
84
+ class ScoreBar(ProtocolModel):
85
+ """One score, and the text a renderer draws it as.
86
+
87
+ ``display`` is computed once, in the builder, so that the terminal and a
88
+ phone message draw the same bar from the same string rather than each
89
+ inventing a scale.
90
+ """
91
+
92
+ label: NonEmptyString
93
+ value: float
94
+ maximum: float = Field(gt=0.0)
95
+ display: NonEmptyString
96
+
97
+
98
+ class TaskResultRow(ProtocolModel):
99
+ """What one committed task contributed to the comparison."""
100
+
101
+ position: int = Field(ge=0)
102
+ task_label: NonEmptyString
103
+ baseline_score: float
104
+ candidate_score: float
105
+ delta: float
106
+ outcome: TaskOutcome
107
+
108
+
109
+ class TaskDisplay(StrEnum):
110
+ """Which task rows a reader asked to see."""
111
+
112
+ ALL = "all"
113
+ CHANGED = "changed"
114
+ REGRESSIONS = "regressions"
115
+ NONE = "none"
116
+
117
+
118
+ #: Losses first, then wins, then ties. Within a group, committed task order.
119
+ _OUTCOME_RANK: Final[dict[str, int]] = {"loss": 0, "win": 1, "tie": 2}
120
+
121
+
122
+ def selected_task_rows(
123
+ rows: list[TaskResultRow], show: TaskDisplay
124
+ ) -> list[TaskResultRow]:
125
+ """Return the rows a reader asked for, worst first.
126
+
127
+ A reader scanning a table wants the rows that moved the wrong way, then the
128
+ ones that moved, then the rest, so the order is fixed here rather than left
129
+ to whichever channel is drawing. ``TaskDisplay.NONE`` selects nothing at
130
+ all, which is the honest reading of a reader who said they did not want the
131
+ table: a channel given no rows prints no table and no heading over it.
132
+
133
+ Selecting rows is all this does. No filter can change a count, because
134
+ every count a reader sees comes from the payload rather than from what a
135
+ channel happened to have room for.
136
+ """
137
+ if show is TaskDisplay.NONE:
138
+ return []
139
+ if show is TaskDisplay.REGRESSIONS:
140
+ chosen = [row for row in rows if row.outcome == "loss"]
141
+ elif show is TaskDisplay.CHANGED:
142
+ chosen = [row for row in rows if row.outcome != "tie"]
143
+ else:
144
+ chosen = list(rows)
145
+ return sorted(chosen, key=lambda row: (_OUTCOME_RANK[row.outcome], row.position))
146
+
147
+
148
+ class SkillSummary(ProtocolModel):
149
+ """One side's Skill, described by size and content address.
150
+
151
+ A baseline with no Skill is a real state rather than a missing value: it is
152
+ what a Skill-insertion comparison measures against, so it has a label and
153
+ no digest.
154
+ """
155
+
156
+ label: NonEmptyString
157
+ root_digest: Digest | None
158
+ file_count: int = Field(ge=0)
159
+ total_bytes: int = Field(ge=0)
160
+
161
+
162
+ class DerivedCost(ProtocolModel):
163
+ """A dollar figure worked out while rendering, from what the run recorded.
164
+
165
+ Decisions document 0007 R6 forbids exactly one thing about cost: a figure
166
+ presented as better sourced than it is. This is not a bill and is never
167
+ drawn as one, so everything it rests on travels with it — the two token
168
+ counts that were multiplied, the prices they were multiplied by, and the
169
+ day those prices were read.
170
+
171
+ ``cached_input_tokens`` and ``prices_name_a_cached_rate`` are carried
172
+ together because a provider that serves part of the prompt from its own
173
+ cache usually charges less for it. When the recorded prices name no cached
174
+ rate, every token is priced at the full rate and the reader is told the
175
+ figure is on the high side, which is the only direction an unstated
176
+ discount can move it. The count is ``None`` when the run recorded no
177
+ usable cache split, which is not the same as a run that cached nothing.
178
+ """
179
+
180
+ usd: float = Field(ge=0.0)
181
+ input_tokens: int = Field(ge=0)
182
+ output_tokens: int = Field(ge=0)
183
+ cached_input_tokens: int | None = Field(default=None, ge=0)
184
+ prices_name_a_cached_rate: bool
185
+ model_id: NonEmptyString
186
+ input_usd_per_mtok: float = Field(gt=0.0)
187
+ output_usd_per_mtok: float = Field(gt=0.0)
188
+ prices_recorded_on: NonEmptyString
189
+
190
+
191
+ class PresentationCaveat(ProtocolModel):
192
+ """One thing a reader must know before believing what they just read.
193
+
194
+ Caveats are part of the payload rather than of a renderer, so that a
195
+ channel cannot drop one by being short of room.
196
+ """
197
+
198
+ code: NonEmptyString
199
+ severity: Literal["info", "warning", "error"]
200
+ text: NonEmptyString
201
+
202
+
203
+ class UpliftPresentationPayload(ProtocolModel):
204
+ """One comparison, ready to be shown anywhere.
205
+
206
+ ``comparison_label`` names which result in the chain this is;
207
+ ``change_label`` names the one thing that differed between the two sides,
208
+ in the arrow form decisions document 0019 section 1 fixes. They are two
209
+ fields because they answer two questions — which receipt am I holding, and
210
+ what did it measure — and a channel with room for only one should not have
211
+ to guess which.
212
+ """
213
+
214
+ schema_version: Literal["techtree.presentation.uplift.v1"]
215
+ run_id: NonEmptyString
216
+ campaign_title: NonEmptyString
217
+ comparison_label: NonEmptyString
218
+ change_label: NonEmptyString
219
+ baseline_skill: SkillSummary
220
+ candidate_skill: SkillSummary
221
+ baseline_score: float
222
+ candidate_score: float
223
+ absolute_delta: float
224
+ relative_delta: float | None
225
+ wins: int = Field(ge=0)
226
+ losses: int = Field(ge=0)
227
+ ties: int = Field(ge=0)
228
+ task_rows: list[TaskResultRow]
229
+ baseline_tasks_scored_full: int | None
230
+ candidate_tasks_scored_full: int | None
231
+ baseline_tokens: int | None
232
+ candidate_tokens: int | None
233
+ baseline_seconds: float | None
234
+ candidate_seconds: float | None
235
+ baseline_model_turns: int | None
236
+ candidate_model_turns: int | None
237
+ baseline_rate_limited_calls: int | None
238
+ candidate_rate_limited_calls: int | None
239
+ every_rollout_completed: bool | None
240
+ economics_source: EconomicsSource
241
+ cost_usd: float | None = Field(default=None, ge=0.0)
242
+ cost_provenance: CostProvenance
243
+ derived_cost: DerivedCost | None = None
244
+ cost_unavailable_reason: NonEmptyString | None = None
245
+ decision: NonEmptyString
246
+ proof_grade: NonEmptyString
247
+ verification_status: NonEmptyString
248
+ caveats: list[PresentationCaveat]
249
+ next_actions: list[NextAction]
250
+
251
+ @model_validator(mode="after")
252
+ def _check_the_rows_and_the_counts_describe_one_comparison(self) -> Self:
253
+ """Reject a payload whose table and headline disagree."""
254
+ outcomes = [row.outcome for row in self.task_rows]
255
+ counts: tuple[tuple[TaskOutcome, int], ...] = (
256
+ ("win", self.wins),
257
+ ("loss", self.losses),
258
+ ("tie", self.ties),
259
+ )
260
+ for outcome, count in counts:
261
+ if outcomes.count(outcome) != count:
262
+ raise ValueError(
263
+ f"the payload reports {count} {outcome} rows and carries "
264
+ f"{outcomes.count(outcome)}"
265
+ )
266
+ positions = [row.position for row in self.task_rows]
267
+ if positions != sorted(positions) or len(set(positions)) != len(positions):
268
+ raise ValueError(
269
+ "task rows are carried in committed task order, each position once"
270
+ )
271
+ return self
272
+
273
+ @model_validator(mode="after")
274
+ def _check_the_cost_is_never_shown_without_its_source(self) -> Self:
275
+ """Reject a payload whose cost claims a provenance it does not have.
276
+
277
+ Decisions document 0007 R6 forbids exactly one thing about cost: a
278
+ figure presented as better sourced than it is. The shape enforces the
279
+ pair here so that no renderer has to remember to.
280
+ """
281
+ known = self.cost_usd is not None
282
+ claims_source = self.cost_provenance is not CostProvenance.UNAVAILABLE
283
+ if known != claims_source:
284
+ raise ValueError(
285
+ "a cost figure needs a provenance and a provenance needs a "
286
+ f"figure; got {self.cost_usd!r} as {self.cost_provenance.value}"
287
+ )
288
+ if known and self.economics_source != "comparison_execution_record":
289
+ raise ValueError(
290
+ "a cost figure comes from the signed execution record; a "
291
+ f"payload sourced from {self.economics_source} has none"
292
+ )
293
+ return self
294
+
295
+ @model_validator(mode="after")
296
+ def _check_a_derived_cost_never_stands_beside_a_reported_one(self) -> Self:
297
+ """Reject a payload carrying two answers to "what did this cost?".
298
+
299
+ A figure the provider reported is the better answer wherever there is
300
+ one, so a derived figure exists only in its absence. Two of them in one
301
+ payload would leave each channel free to pick, and two channels showing
302
+ one run would then be able to disagree about money.
303
+ """
304
+ if self.derived_cost is not None and self.cost_usd is not None:
305
+ raise ValueError(
306
+ "a cost is derived only when none was reported; this payload "
307
+ f"carries both {self.derived_cost.usd} and {self.cost_usd}"
308
+ )
309
+ figure = self.derived_cost is not None or self.cost_usd is not None
310
+ if figure == (self.cost_unavailable_reason is not None):
311
+ raise ValueError(
312
+ "a payload with no cost figure says what is missing, and one "
313
+ "with a figure has nothing to explain away"
314
+ )
315
+ return self
316
+
317
+ @model_validator(mode="after")
318
+ def _check_the_counts_read_from_the_run_arrive_together(self) -> Self:
319
+ """Reject a payload that read half of one run's recorded traces.
320
+
321
+ Turns, throttling and whether every rollout finished are one reading of
322
+ one pair of recorded, digest-checked files. A payload holding some of
323
+ them and not the others would be describing a reading that never
324
+ happened.
325
+ """
326
+ read = (
327
+ self.baseline_model_turns,
328
+ self.candidate_model_turns,
329
+ self.baseline_rate_limited_calls,
330
+ self.candidate_rate_limited_calls,
331
+ self.every_rollout_completed,
332
+ )
333
+ if None in read and any(value is not None for value in read):
334
+ raise ValueError(
335
+ "the counts read from a run's recorded traces are all present "
336
+ f"or all absent; got {read}"
337
+ )
338
+ return self
339
+
340
+ @model_validator(mode="after")
341
+ def _check_the_task_counts_fit_the_table(self) -> Self:
342
+ """Reject a headline count that no per-task table could produce."""
343
+ baseline = self.baseline_tasks_scored_full
344
+ candidate = self.candidate_tasks_scored_full
345
+ if baseline is None or candidate is None:
346
+ if baseline is not candidate:
347
+ raise ValueError(
348
+ "both sides carry a task count or neither does; got "
349
+ f"{(baseline, candidate)}"
350
+ )
351
+ return self
352
+ for count in (baseline, candidate):
353
+ if not 0 <= count <= len(self.task_rows):
354
+ raise ValueError(
355
+ f"a side scored between 0 and {len(self.task_rows)} of the "
356
+ f"comparison's tasks; got {count}"
357
+ )
358
+ return self
@@ -0,0 +1,312 @@
1
+ """The terminal rendering. Spec section 7.15.
2
+
3
+ This is the CLI's own renderer, and in the released product it is the whole
4
+ terminal result path: a signed report becomes a neutral payload, and this draws
5
+ it. No model is asked to explain a result, so nothing between the numbers and
6
+ the reader can change one by drawing it.
7
+
8
+ Four rules, and each one is about a reader rather than about a terminal.
9
+
10
+ *Meaning is never carried by colour alone.* Every outcome is spelled ``WIN``,
11
+ ``LOSS`` or ``TIE``, every caveat is introduced by its severity in words, and a
12
+ terminal with no colour at all loses nothing but decoration. ``NO_COLOR`` and
13
+ ``--no-color`` are honoured by the console the CLI builds, and Rich emits no
14
+ escape sequences at all when stdout is not a terminal.
15
+
16
+ *The rendering is deterministic.* No spinner, no progress animation, no clock,
17
+ no dictionary iteration order: the same payload renders to the same bytes,
18
+ which is what makes it testable at all.
19
+
20
+ *The order is the order of trust.* What was compared, then the result, then the
21
+ tasks, then what changed, then the caveats — so a reader who stops early stops
22
+ having read the honest version. The caveats are never last-resort small print
23
+ they can miss; a development-only or failed-verification result says so in the
24
+ badge at the top as well.
25
+
26
+ *Regressions come first.* A reader scanning a table wants the rows that moved
27
+ the wrong way, then the ones that moved, then the rest.
28
+
29
+ The payload's next actions are deliberately not drawn here. Every Techtree
30
+ command ends with the same numbered next-steps block, rendered by the CLI from
31
+ the envelope it returns, and a result that grew a second one of its own would
32
+ be the only command in the product that answers that question twice.
33
+ """
34
+
35
+ from __future__ import annotations
36
+
37
+ from typing import Final
38
+
39
+ from rich.console import Console
40
+ from rich.table import Table
41
+
42
+ from techtree.presentation.build import (
43
+ HELD_FIXED_LINE,
44
+ NOT_BROAD_CAPABILITY_LINE,
45
+ VERIFICATION_FAILED,
46
+ VERIFICATION_NOT_VERIFIED,
47
+ VERIFICATION_VERIFIED,
48
+ cost_explanation,
49
+ cost_summary,
50
+ decision_headline,
51
+ efficiency_sentence,
52
+ score_bars,
53
+ solved_line,
54
+ task_count_line,
55
+ )
56
+ from techtree.presentation.models import (
57
+ TaskDisplay,
58
+ UpliftPresentationPayload,
59
+ selected_task_rows,
60
+ )
61
+
62
+ __all__ = ["outcome_label", "render_uplift_console"]
63
+
64
+ _SEVERITY_PREFIX: Final[dict[str, str]] = {
65
+ "info": "Note:",
66
+ "warning": "Warning:",
67
+ "error": "Error:",
68
+ }
69
+
70
+ _SEVERITY_STYLE: Final[dict[str, str]] = {
71
+ "info": "",
72
+ "warning": "yellow",
73
+ "error": "red",
74
+ }
75
+
76
+ _OUTCOME_LABEL: Final[dict[str, str]] = {
77
+ "win": "WIN",
78
+ "loss": "LOSS",
79
+ "tie": "TIE",
80
+ }
81
+
82
+ _VERIFICATION_BADGE: Final[dict[str, str]] = {
83
+ VERIFICATION_VERIFIED: "local proof verified offline",
84
+ VERIFICATION_FAILED: "LOCAL PROOF DID NOT VERIFY",
85
+ VERIFICATION_NOT_VERIFIED: "local proof not checked",
86
+ }
87
+
88
+
89
+ def outcome_label(outcome: str) -> str:
90
+ """Return the word a reader sees for one task's outcome."""
91
+ return _OUTCOME_LABEL[outcome]
92
+
93
+
94
+ def render_uplift_console(
95
+ payload: UpliftPresentationPayload,
96
+ console: Console,
97
+ *,
98
+ show_tasks: TaskDisplay = TaskDisplay.CHANGED,
99
+ ) -> None:
100
+ """Render an accessible, side-by-side terminal result.
101
+
102
+ ``show_tasks`` is the reader's choice of how much of the per-task table to
103
+ print (spec section 7.21). It selects rows and nothing else: no filter here
104
+ can change a count, because the counts come from the payload.
105
+ """
106
+ _header(payload, console)
107
+ _primary(payload, console)
108
+ _tasks(payload, console, show_tasks)
109
+ _efficiency(payload, console)
110
+ _what_changed(payload, console)
111
+ _caveats(payload, console)
112
+
113
+
114
+ def _header(payload: UpliftPresentationPayload, console: Console) -> None:
115
+ """Campaign, comparison, and the badge that says how much this is worth."""
116
+ console.print(payload.campaign_title)
117
+ console.print(payload.comparison_label)
118
+ console.print(
119
+ f"[{payload.proof_grade} · {_VERIFICATION_BADGE[payload.verification_status]}]"
120
+ )
121
+ console.print()
122
+
123
+
124
+ def _primary(payload: UpliftPresentationPayload, console: Console) -> None:
125
+ """What was established, what is still failing, and the numbers under it.
126
+
127
+ The three lines at the top are the ones a reader repeats to somebody else,
128
+ so they are the three that have to be defensible on their own: what this
129
+ comparison showed, how much of the task family is still unsolved, and the
130
+ fact that none of it is evidence about broad capability.
131
+
132
+ Wins, losses and ties do not appear here. On this Climb the baseline scores
133
+ nothing on every task, so a tie means the candidate failed the task too,
134
+ and a headline of twenty-four wins and no losses would read as a clean
135
+ sweep over a run with twelve tasks still failing. The three words stay in
136
+ the per-task table below, where each one is beside the scores that produced
137
+ it.
138
+
139
+ The count leads the detail because it is the number a person reads a result
140
+ in. A mean of 0.667 and "24 of 36" are the same measurement, and only one
141
+ of them can be repeated to somebody over a table.
142
+ """
143
+ console.print(f"Result {decision_headline(payload)}")
144
+ console.print(f" {solved_line(payload)}")
145
+ console.print(f" {NOT_BROAD_CAPABILITY_LINE}")
146
+ console.print()
147
+
148
+ counted = task_count_line(payload)
149
+ if counted is not None:
150
+ console.print(f"Tasks {counted}")
151
+ console.print()
152
+
153
+ table = Table(box=None, show_header=False, pad_edge=False, padding=(0, 2))
154
+ table.add_column("side", no_wrap=True)
155
+ table.add_column("bar", no_wrap=True)
156
+ for bar in score_bars(payload):
157
+ table.add_row(bar.label, bar.display)
158
+ console.print(table)
159
+ console.print()
160
+
161
+ console.print(
162
+ f"Change {payload.absolute_delta:+.3f} mean score ({_relative(payload)})"
163
+ )
164
+ console.print()
165
+
166
+
167
+ def _relative(payload: UpliftPresentationPayload) -> str:
168
+ """Say what a relative change is, or why there is not one."""
169
+ if payload.relative_delta is None:
170
+ return "no relative change: the baseline scored nothing"
171
+ return f"{payload.relative_delta:+.1%} relative"
172
+
173
+
174
+ def _tasks(
175
+ payload: UpliftPresentationPayload, console: Console, show: TaskDisplay
176
+ ) -> None:
177
+ """The per-task table, regressions first."""
178
+ if show is TaskDisplay.NONE:
179
+ return
180
+ rows = selected_task_rows(payload.task_rows, show)
181
+ if not rows:
182
+ console.print(_nothing_selected(show))
183
+ console.print()
184
+ return
185
+
186
+ table = Table(box=None, pad_edge=False, padding=(0, 2))
187
+ table.add_column("Task", no_wrap=True)
188
+ table.add_column("Baseline", justify="right", no_wrap=True)
189
+ table.add_column("Candidate", justify="right", no_wrap=True)
190
+ table.add_column("Change", justify="right", no_wrap=True)
191
+ table.add_column("Outcome", no_wrap=True)
192
+ for row in rows:
193
+ table.add_row(
194
+ row.task_label,
195
+ f"{row.baseline_score:.3f}",
196
+ f"{row.candidate_score:.3f}",
197
+ f"{row.delta:+.3f}",
198
+ outcome_label(row.outcome),
199
+ )
200
+ console.print(table)
201
+ if len(rows) != len(payload.task_rows):
202
+ console.print(
203
+ f"Showing {len(rows)} of {len(payload.task_rows)} tasks. "
204
+ "Use --show-tasks all for the rest."
205
+ )
206
+ console.print()
207
+
208
+
209
+ def _nothing_selected(show: TaskDisplay) -> str:
210
+ """Say why the table is empty, in terms of what was asked for.
211
+
212
+ An empty selection and an unchanged comparison are not the same fact, and
213
+ one sentence cannot honestly stand for both. Asking a run that solved
214
+ twenty-three tasks to list its regressions finds none, and answering that
215
+ with "every task scored the same" contradicts the three lines directly
216
+ above it — the reader is told the run improved and then told it did not.
217
+ So each selection says what its own emptiness means.
218
+ """
219
+ if show is TaskDisplay.REGRESSIONS:
220
+ return "No task scored worse with the Skill than without it."
221
+ if show is TaskDisplay.ALL:
222
+ return "This run recorded no tasks."
223
+ return "Every task scored the same on both sides."
224
+
225
+
226
+ def _efficiency(payload: UpliftPresentationPayload, console: Console) -> None:
227
+ """Turns, tokens, time and cost, each shown with the source it came from.
228
+
229
+ Decisions document 0007 R6 governs the cost line: a figure is never printed
230
+ without saying where it is from, because "$4.10" and "$4.10, estimated" are
231
+ different claims and only one of them is about money that was actually
232
+ charged.
233
+
234
+ The sentence under the numbers is there because two durations side by side
235
+ are not a finding. What the two sides did differently is legible in how
236
+ many times each had to go back to the model, and that is the half of the
237
+ comparison a different machine would reproduce.
238
+ """
239
+ seconds = _pair(payload.baseline_seconds, payload.candidate_seconds, unit="s")
240
+ console.print("Efficiency")
241
+ turns = _pair(payload.baseline_model_turns, payload.candidate_model_turns)
242
+ console.print(f" Turns {turns}")
243
+ console.print(
244
+ f" Tokens {_pair(payload.baseline_tokens, payload.candidate_tokens)}"
245
+ )
246
+ console.print(f" Time {seconds}")
247
+ console.print(f" Cost {cost_summary(payload)}")
248
+ for line in cost_explanation(payload):
249
+ console.print(f" {line}")
250
+ console.print(f" Source {_ECONOMICS_SOURCE[payload.economics_source]}")
251
+ sentence = efficiency_sentence(payload)
252
+ if sentence is not None:
253
+ console.print(f" {sentence}")
254
+ console.print()
255
+
256
+
257
+ def _pair(
258
+ baseline: float | int | None, candidate: float | int | None, *, unit: str = ""
259
+ ) -> str:
260
+ """Return both sides of one measurement, or say it was not recorded."""
261
+ if baseline is None and candidate is None:
262
+ return "not recorded for this run"
263
+ return f"baseline {_number(baseline, unit)}, candidate {_number(candidate, unit)}"
264
+
265
+
266
+ def _number(value: float | int | None, unit: str) -> str:
267
+ """Return one measurement, or the word for an absent one.
268
+
269
+ Whole counts are grouped. A token total is seven digits in a real run, and
270
+ seven ungrouped digits are a number a reader has to count rather than read.
271
+ """
272
+ if value is None:
273
+ return "unavailable"
274
+ if isinstance(value, float):
275
+ return f"{value:.1f}{unit}"
276
+ return f"{value:,}{unit}"
277
+
278
+
279
+ def _what_changed(payload: UpliftPresentationPayload, console: Console) -> None:
280
+ """The one declared difference, and the statement that it is the only one."""
281
+ console.print("What changed")
282
+ console.print(f" {payload.change_label}")
283
+ for side, skill in (
284
+ ("Baseline ", payload.baseline_skill),
285
+ ("Candidate", payload.candidate_skill),
286
+ ):
287
+ digest = "none" if skill.root_digest is None else skill.root_digest
288
+ console.print(f" {side} {skill.label}")
289
+ console.print(f" {digest}")
290
+ console.print(f" {HELD_FIXED_LINE}")
291
+ console.print(" Each of those was checked against what the run actually did.")
292
+ console.print()
293
+
294
+
295
+ #: Where the numbers above came from, said in the reader's terms.
296
+ _ECONOMICS_SOURCE: Final[dict[str, str]] = {
297
+ "comparison_execution_record": "this run's signed execution record",
298
+ "episode_receipts": "the run's receipts; no execution record was written",
299
+ "unavailable": "nothing recorded it",
300
+ }
301
+
302
+
303
+ def _caveats(payload: UpliftPresentationPayload, console: Console) -> None:
304
+ """Every caveat, in payload order, introduced by severity in words."""
305
+ if not payload.caveats:
306
+ return
307
+ console.print("What this does and does not prove")
308
+ for caveat in payload.caveats:
309
+ console.print(
310
+ f" {_SEVERITY_PREFIX[caveat.severity]} {caveat.text}",
311
+ style=_SEVERITY_STYLE[caveat.severity] or None,
312
+ )