techtree 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- techtree/__init__.py +35 -0
- techtree/__main__.py +14 -0
- techtree/canonical.py +239 -0
- techtree/catalog/__init__.py +25 -0
- techtree/catalog/repository.py +400 -0
- techtree/catalog/service.py +419 -0
- techtree/cli/__init__.py +1 -0
- techtree/cli/app.py +416 -0
- techtree/cli/commands/__init__.py +1 -0
- techtree/cli/commands/climb.py +1223 -0
- techtree/cli/commands/doctor.py +147 -0
- techtree/cli/commands/engine.py +207 -0
- techtree/cli/commands/proof.py +556 -0
- techtree/cli/commands/publish.py +447 -0
- techtree/cli/commands/release.py +303 -0
- techtree/cli/commands/run.py +1067 -0
- techtree/cli/commands/setup.py +181 -0
- techtree/cli/commands/skill.py +221 -0
- techtree/cli/commands/uplift.py +698 -0
- techtree/cli/commands/withdraw.py +212 -0
- techtree/cli/confirm.py +47 -0
- techtree/cli/context.py +96 -0
- techtree/cli/invoke.py +220 -0
- techtree/cli/output.py +280 -0
- techtree/constants.py +138 -0
- techtree/crypto.py +128 -0
- techtree/doctor/__init__.py +1 -0
- techtree/doctor/checks.py +675 -0
- techtree/doctor/execution_checks.py +435 -0
- techtree/doctor/service.py +326 -0
- techtree/drafts/__init__.py +32 -0
- techtree/drafts/source.py +146 -0
- techtree/drafts/store.py +992 -0
- techtree/engines/__init__.py +1 -0
- techtree/engines/bundle.py +251 -0
- techtree/engines/installer.py +679 -0
- techtree/engines/registry.py +235 -0
- techtree/engines/runner.py +170 -0
- techtree/errors.py +262 -0
- techtree/fs.py +234 -0
- techtree/harness.py +108 -0
- techtree/identity/__init__.py +41 -0
- techtree/identity/models.py +113 -0
- techtree/identity/service.py +199 -0
- techtree/identity/store.py +263 -0
- techtree/ids.py +85 -0
- techtree/manifests/__init__.py +39 -0
- techtree/manifests/builder.py +433 -0
- techtree/manifests/compare.py +376 -0
- techtree/models/__init__.py +282 -0
- techtree/models/base.py +201 -0
- techtree/models/campaign.py +484 -0
- techtree/models/catalog.py +227 -0
- techtree/models/cli.py +151 -0
- techtree/models/climb.py +254 -0
- techtree/models/data_policy.py +130 -0
- techtree/models/engine.py +156 -0
- techtree/models/episode_receipt.py +130 -0
- techtree/models/evaluation_backend.py +113 -0
- techtree/models/experiment.py +154 -0
- techtree/models/run.py +214 -0
- techtree/models/skill.py +156 -0
- techtree/models/uplift_report.py +158 -0
- techtree/models/validation.py +299 -0
- techtree/paths.py +116 -0
- techtree/presentation/__init__.py +31 -0
- techtree/presentation/build.py +1242 -0
- techtree/presentation/compact.py +246 -0
- techtree/presentation/evidence.py +169 -0
- techtree/presentation/models.py +358 -0
- techtree/presentation/rich.py +312 -0
- techtree/presentation/sanitize.py +156 -0
- techtree/publication/__init__.py +44 -0
- techtree/publication/address.py +180 -0
- techtree/publication/coordinates.py +26 -0
- techtree/publication/journal.py +212 -0
- techtree/publication/keccak.py +183 -0
- techtree/publication/models.py +209 -0
- techtree/publication/offer.py +35 -0
- techtree/publication/service.py +618 -0
- techtree/publication/transport.py +296 -0
- techtree/publication/verify.py +242 -0
- techtree/publication/withdraw.py +156 -0
- techtree/py.typed +0 -0
- techtree/receipts/__init__.py +52 -0
- techtree/receipts/bundle.py +578 -0
- techtree/receipts/compare.py +1065 -0
- techtree/receipts/episode.py +672 -0
- techtree/receipts/execution.py +630 -0
- techtree/receipts/observed.py +474 -0
- techtree/receipts/set.py +336 -0
- techtree/receipts/uplift.py +655 -0
- techtree/receipts/verify.py +1055 -0
- techtree/release/__init__.py +9 -0
- techtree/release/bootstrap.py +509 -0
- techtree/release/checks.py +376 -0
- techtree/release/document.py +125 -0
- techtree/release/generate.py +221 -0
- techtree/release/models.py +293 -0
- techtree/release/provenance.py +109 -0
- techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
- techtree/resources/catalog/catalog.json +32 -0
- techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
- techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
- techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
- techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
- techtree/resources/engines/default/engine.json +20 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
- techtree/resources/engines/default/pyproject.toml +23 -0
- techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
- techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
- techtree/resources/engines/default/tools/normalize_validation.py +222 -0
- techtree/resources/engines/default/uv.lock +1758 -0
- techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
- techtree/resources/release/build-provenance.json +4 -0
- techtree/resources/release/release-core.json +24 -0
- techtree/runs/__init__.py +31 -0
- techtree/runs/artifacts.py +750 -0
- techtree/runs/child_registry.py +228 -0
- techtree/runs/events.py +478 -0
- techtree/runs/executor.py +140 -0
- techtree/runs/fake.py +741 -0
- techtree/runs/launcher.py +253 -0
- techtree/runs/machine.py +489 -0
- techtree/runs/real.py +789 -0
- techtree/runs/service.py +616 -0
- techtree/runs/store.py +555 -0
- techtree/runs/validation.py +259 -0
- techtree/runs/variants.py +684 -0
- techtree/settings.py +143 -0
- techtree/skills/__init__.py +14 -0
- techtree/skills/archive.py +282 -0
- techtree/skills/policy.py +62 -0
- techtree/skills/scanner.py +394 -0
- techtree/skills/service.py +752 -0
- techtree/skills/starter.py +434 -0
- techtree/tasksets/__init__.py +1 -0
- techtree/tasksets/membership.py +269 -0
- techtree/tasksets/provider.py +207 -0
- techtree/tasksets/resolver.py +311 -0
- techtree/tasksets/service.py +484 -0
- techtree/tasksets/verifiers_cli.py +538 -0
- techtree/uplift/__init__.py +20 -0
- techtree/uplift/context.py +544 -0
- techtree/uplift/derive.py +203 -0
- techtree/uplift/public_tasks.py +151 -0
- techtree/uplift/service.py +719 -0
- techtree/uplift/source.py +160 -0
- techtree/verifiers/__init__.py +31 -0
- techtree/verifiers/budget.py +219 -0
- techtree/verifiers/child.py +633 -0
- techtree/verifiers/compiler.py +432 -0
- techtree/verifiers/config.py +365 -0
- techtree/verifiers/credentials.py +321 -0
- techtree/verifiers/image.py +126 -0
- techtree/verifiers/models.py +527 -0
- techtree/verifiers/outputs.py +368 -0
- techtree/verifiers/progress.py +192 -0
- techtree/verifiers/supervisor.py +341 -0
- techtree/verifiers/verify.py +782 -0
- techtree/version.py +39 -0
- techtree/worker/__init__.py +18 -0
- techtree/worker/execute.py +487 -0
- techtree/worker/main.py +57 -0
- techtree-0.1.0.dist-info/METADATA +344 -0
- techtree-0.1.0.dist-info/RECORD +174 -0
- techtree-0.1.0.dist-info/WHEEL +4 -0
- techtree-0.1.0.dist-info/entry_points.txt +3 -0
- techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
"""The rendering a phone can carry. Spec section 7.16.
|
|
2
|
+
|
|
3
|
+
A gateway message is not a small terminal. It has no escape sequences, no
|
|
4
|
+
column alignment, a reader holding a phone, and a channel that may truncate
|
|
5
|
+
anything long. So this renderer is bounded by construction: a headline, the
|
|
6
|
+
counts, the honest qualifications, at most a few task rows, and one sentence
|
|
7
|
+
about what could happen next.
|
|
8
|
+
|
|
9
|
+
The row cap is a default, not a refusal. A reader who says ``--show-tasks all``
|
|
10
|
+
has asked for every task on purpose, and the bound exists so that an unasked-for
|
|
11
|
+
message is short rather than so that somebody who asked can be told no. So the
|
|
12
|
+
cap applies to the selections a reader did not have to name, and the explicit
|
|
13
|
+
request for everything overrides it.
|
|
14
|
+
|
|
15
|
+
Two things are *not* dropped to make it fit.
|
|
16
|
+
|
|
17
|
+
*Every warning and error survives.* Room is made by cutting the table, never by
|
|
18
|
+
cutting a caveat. A result that only says what went well on a phone and keeps
|
|
19
|
+
its qualifications for the desktop would be dishonest in exactly the channel
|
|
20
|
+
where somebody is most likely to forward it to someone else.
|
|
21
|
+
|
|
22
|
+
*The proof grade travels with the numbers.* A one-line quote of an uplift with
|
|
23
|
+
no grade beside it is the shape misinformation takes, so the grade and whether
|
|
24
|
+
its proof verified are part of the same block as the scores.
|
|
25
|
+
|
|
26
|
+
Nothing here emits ANSI, and nothing here formats a number differently from the
|
|
27
|
+
terminal renderer: both read the same payload fields.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
from typing import Final
|
|
33
|
+
|
|
34
|
+
from techtree.presentation.build import (
|
|
35
|
+
HELD_FIXED_LINE,
|
|
36
|
+
NOT_BROAD_CAPABILITY_LINE,
|
|
37
|
+
VERIFICATION_FAILED,
|
|
38
|
+
VERIFICATION_NOT_VERIFIED,
|
|
39
|
+
VERIFICATION_VERIFIED,
|
|
40
|
+
cost_explanation,
|
|
41
|
+
cost_summary,
|
|
42
|
+
decision_headline,
|
|
43
|
+
efficiency_sentence,
|
|
44
|
+
solved_line,
|
|
45
|
+
task_count_line,
|
|
46
|
+
)
|
|
47
|
+
from techtree.presentation.models import (
|
|
48
|
+
TaskDisplay,
|
|
49
|
+
UpliftPresentationPayload,
|
|
50
|
+
selected_task_rows,
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
__all__ = [
|
|
54
|
+
"DEFAULT_MAXIMUM_TASK_ROWS",
|
|
55
|
+
"UNVERIFIED_HEADLINE",
|
|
56
|
+
"render_uplift_markdown",
|
|
57
|
+
]
|
|
58
|
+
|
|
59
|
+
#: Enough rows to show a pattern, few enough to read on a phone.
|
|
60
|
+
DEFAULT_MAXIMUM_TASK_ROWS: Final = 5
|
|
61
|
+
|
|
62
|
+
#: What a result whose proof failed says before it says anything else.
|
|
63
|
+
UNVERIFIED_HEADLINE: Final = (
|
|
64
|
+
"**This run's local proof did not verify. Do not rely on the numbers below.**"
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
_VERIFICATION_PHRASE: Final[dict[str, str]] = {
|
|
68
|
+
VERIFICATION_VERIFIED: "signature verified offline",
|
|
69
|
+
VERIFICATION_FAILED: "signature DID NOT verify",
|
|
70
|
+
VERIFICATION_NOT_VERIFIED: "signature not checked",
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def render_uplift_markdown(
|
|
75
|
+
payload: UpliftPresentationPayload,
|
|
76
|
+
*,
|
|
77
|
+
maximum_task_rows: int = DEFAULT_MAXIMUM_TASK_ROWS,
|
|
78
|
+
show_tasks: TaskDisplay = TaskDisplay.CHANGED,
|
|
79
|
+
) -> str:
|
|
80
|
+
"""Return compact Markdown suitable for a phone or gateway message.
|
|
81
|
+
|
|
82
|
+
A result whose proof did not verify says that first and in bold. This is
|
|
83
|
+
the channel a number is most likely to be quoted out of, so the sentence
|
|
84
|
+
that would stop somebody quoting it cannot be further down.
|
|
85
|
+
|
|
86
|
+
``show_tasks`` is the same choice the terminal renderer takes, answered by
|
|
87
|
+
the same reader through the same option, and it selects rows through the
|
|
88
|
+
same rule. A reader who asks a piped command for every task and is handed
|
|
89
|
+
the default five has been told the option works when it did not.
|
|
90
|
+
"""
|
|
91
|
+
lines = [
|
|
92
|
+
f"**{decision_headline(payload)} — {solved_line(payload)}**",
|
|
93
|
+
"",
|
|
94
|
+
f"- {NOT_BROAD_CAPABILITY_LINE}",
|
|
95
|
+
]
|
|
96
|
+
if payload.verification_status == VERIFICATION_FAILED:
|
|
97
|
+
lines = [UNVERIFIED_HEADLINE, "", *lines]
|
|
98
|
+
lines += [
|
|
99
|
+
f"- {payload.campaign_title} — {payload.comparison_label}",
|
|
100
|
+
f"- Changed: {payload.change_label}. {HELD_FIXED_LINE}",
|
|
101
|
+
f"- Tasks: {_headline_numbers(payload)}",
|
|
102
|
+
f"- Proof: local {payload.proof_grade}, "
|
|
103
|
+
f"{_VERIFICATION_PHRASE[payload.verification_status]}",
|
|
104
|
+
f"- Cost: {cost_summary(payload)}",
|
|
105
|
+
*(f"- {line}" for line in cost_explanation(payload)),
|
|
106
|
+
*_work(payload),
|
|
107
|
+
"- Raw episodes: retained locally; not uploaded",
|
|
108
|
+
]
|
|
109
|
+
|
|
110
|
+
lines += _table(payload, show_tasks, maximum_task_rows)
|
|
111
|
+
|
|
112
|
+
qualifications = [caveat for caveat in payload.caveats if caveat.severity != "info"]
|
|
113
|
+
if qualifications:
|
|
114
|
+
lines.append("")
|
|
115
|
+
lines.extend(f"- {caveat.text}" for caveat in qualifications)
|
|
116
|
+
|
|
117
|
+
lines.append("")
|
|
118
|
+
lines.append(_next_line(payload))
|
|
119
|
+
return "\n".join(lines)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _headline_numbers(payload: UpliftPresentationPayload) -> str:
|
|
123
|
+
"""Return how the two sides scored, in the unit a person counts in.
|
|
124
|
+
|
|
125
|
+
The bold line above carries what was established and what is still failing.
|
|
126
|
+
This carries the movement underneath it: both sides' counts where the
|
|
127
|
+
reward has them, and the means they came from in the same breath, so that
|
|
128
|
+
nothing is lost by quoting either one.
|
|
129
|
+
"""
|
|
130
|
+
means = (
|
|
131
|
+
f"{payload.baseline_score:.3f} → {payload.candidate_score:.3f} "
|
|
132
|
+
f"({payload.absolute_delta:+.3f})"
|
|
133
|
+
)
|
|
134
|
+
counted = task_count_line(payload)
|
|
135
|
+
if counted is None:
|
|
136
|
+
return f"mean {means}"
|
|
137
|
+
return f"{counted}, mean {means}"
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _work(payload: UpliftPresentationPayload) -> list[str]:
|
|
141
|
+
"""Return what each side spent doing the same tasks, in one bullet.
|
|
142
|
+
|
|
143
|
+
The channel with the least room still carries this: a Skill that took a
|
|
144
|
+
third of the model turns did something a reader wants to know, and unlike
|
|
145
|
+
the clock, that number does not move when the same run is repeated on a
|
|
146
|
+
busier afternoon. The sentence carries both times, so a run that has it
|
|
147
|
+
does not also get a bare pair of durations.
|
|
148
|
+
"""
|
|
149
|
+
sentence = efficiency_sentence(payload)
|
|
150
|
+
if sentence is not None:
|
|
151
|
+
return [f"- Work: {sentence}"]
|
|
152
|
+
return [f"- Time: {_time(payload)}"]
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _time(payload: UpliftPresentationPayload) -> str:
|
|
156
|
+
"""Return how long each side took, or say it was not recorded.
|
|
157
|
+
|
|
158
|
+
Decisions document 0019 section 3 puts timing in the measured difference,
|
|
159
|
+
so the channel with the least room still carries it: a comparison whose
|
|
160
|
+
two sides took very different amounts of time is a comparison a reader
|
|
161
|
+
should be able to ask about.
|
|
162
|
+
"""
|
|
163
|
+
baseline = payload.baseline_seconds
|
|
164
|
+
candidate = payload.candidate_seconds
|
|
165
|
+
if baseline is None and candidate is None:
|
|
166
|
+
return "not recorded for this run"
|
|
167
|
+
return f"baseline {_seconds(baseline)}, candidate {_seconds(candidate)}"
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _seconds(value: float | None) -> str:
|
|
171
|
+
"""Return one side's elapsed time, or the word for an absent one."""
|
|
172
|
+
if value is None:
|
|
173
|
+
return "unavailable"
|
|
174
|
+
return f"{value:.1f}s"
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _next_line(payload: UpliftPresentationPayload) -> str:
|
|
178
|
+
"""Offer, in one line, the steps this result actually has.
|
|
179
|
+
|
|
180
|
+
A development-only result has no proof to check and nothing may be derived
|
|
181
|
+
from it, so it is offered the one thing it can do. Everything else is left
|
|
182
|
+
unsaid rather than promised on a channel with no room to explain.
|
|
183
|
+
"""
|
|
184
|
+
if payload.proof_grade == "development_only":
|
|
185
|
+
return "Next: I can show every task locally."
|
|
186
|
+
return (
|
|
187
|
+
"Next: I can show every task locally, set up a comparison against a "
|
|
188
|
+
"revised Skill, or check this run's local receipt offline with "
|
|
189
|
+
"`techtree proof verify`."
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _table(
|
|
194
|
+
payload: UpliftPresentationPayload, show: TaskDisplay, maximum_task_rows: int
|
|
195
|
+
) -> list[str]:
|
|
196
|
+
"""Return the heading and the rows for the table this reader asked for.
|
|
197
|
+
|
|
198
|
+
The cap is the bound this channel keeps by default, and it is applied to
|
|
199
|
+
every selection except the one a reader had to type out. Asking for all
|
|
200
|
+
tasks is a deliberate override of a default, so it is honoured; a reader
|
|
201
|
+
who wanted a short message never asked for the long one.
|
|
202
|
+
|
|
203
|
+
A reader who asked for no table, or a selection with nothing in it, gets
|
|
204
|
+
neither rows nor a heading over them. A heading with nothing underneath is
|
|
205
|
+
a line that says a table exists somewhere it does not.
|
|
206
|
+
"""
|
|
207
|
+
selected = selected_task_rows(payload.task_rows, show)
|
|
208
|
+
shown = selected if show is TaskDisplay.ALL else selected[:maximum_task_rows]
|
|
209
|
+
if not shown:
|
|
210
|
+
return []
|
|
211
|
+
return [
|
|
212
|
+
"",
|
|
213
|
+
_heading(payload, show, shown=len(shown), selected=len(selected)),
|
|
214
|
+
*(
|
|
215
|
+
f"- {row.task_label}: {row.baseline_score:.2f} → "
|
|
216
|
+
f"{row.candidate_score:.2f} ({row.outcome.upper()})"
|
|
217
|
+
for row in shown
|
|
218
|
+
),
|
|
219
|
+
]
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _heading(
|
|
223
|
+
payload: UpliftPresentationPayload,
|
|
224
|
+
show: TaskDisplay,
|
|
225
|
+
*,
|
|
226
|
+
shown: int,
|
|
227
|
+
selected: int,
|
|
228
|
+
) -> str:
|
|
229
|
+
"""Say what the rows underneath are, and say when some were left out.
|
|
230
|
+
|
|
231
|
+
The heading a reader can check is the heading that names its own
|
|
232
|
+
selection. Calling a full table the largest changes claims a ranking that
|
|
233
|
+
nothing performed, and calling a list of losses by the same name reads as
|
|
234
|
+
though the wins had also been in the running and only just lost.
|
|
235
|
+
|
|
236
|
+
The counts follow the same rule. Where every row a reader asked for is
|
|
237
|
+
printed, the pair says how much of the whole comparison that is; where the
|
|
238
|
+
cap left some out, it says how many of the asked-for rows are here instead,
|
|
239
|
+
because that is the number that tells a reader something is missing.
|
|
240
|
+
"""
|
|
241
|
+
if show is TaskDisplay.ALL:
|
|
242
|
+
return f"All {selected} tasks:"
|
|
243
|
+
name = "Regressions" if show is TaskDisplay.REGRESSIONS else "Changed tasks"
|
|
244
|
+
if shown < selected:
|
|
245
|
+
return f"{name} ({shown} of {selected} shown):"
|
|
246
|
+
return f"{name} ({selected} of {len(payload.task_rows)} tasks):"
|
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
"""Counts read back out of a finished run's own recorded evidence.
|
|
2
|
+
|
|
3
|
+
Two facts a reader of a result asks for are recorded by every run and carried
|
|
4
|
+
by none of its signed documents: how many model turns each side took, and how
|
|
5
|
+
often the provider refused a model call. The first is the efficiency finding a
|
|
6
|
+
Skill's whole value can sit in; the second is a validity question, because two
|
|
7
|
+
sides that met different amounts of throttling did not meet identical
|
|
8
|
+
conditions even when both finished.
|
|
9
|
+
|
|
10
|
+
Neither is added to a signed artifact here. The signed
|
|
11
|
+
:class:`~techtree.receipts.execution.ComparisonExecutionRecord` already commits
|
|
12
|
+
each side's normalized episodes and raw traces *by digest*, so this reads those
|
|
13
|
+
two files back at render time and checks them against the digests the record
|
|
14
|
+
already holds. A file that is missing, unreadable, or no longer the file the
|
|
15
|
+
record committed yields nothing at all rather than a number: an unchecked count
|
|
16
|
+
would be worth less than saying it is unknown.
|
|
17
|
+
|
|
18
|
+
Only counts leave this module. The files it opens carry prompts, replies and
|
|
19
|
+
grader material, and nothing here returns a string taken from either of them.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import json
|
|
25
|
+
from dataclasses import dataclass
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
from typing import Final
|
|
28
|
+
|
|
29
|
+
from techtree.canonical import sha256_digest_bytes
|
|
30
|
+
from techtree.errors import ValidationError
|
|
31
|
+
from techtree.models.experiment import ExperimentVariant
|
|
32
|
+
from techtree.receipts.execution import ComparisonExecutionRecord
|
|
33
|
+
from techtree.verifiers.models import RunPaths, VariantName
|
|
34
|
+
from techtree.verifiers.outputs import TRACES_FILENAME, read_normalized_episodes
|
|
35
|
+
|
|
36
|
+
__all__ = [
|
|
37
|
+
"RATE_LIMIT_STATUS",
|
|
38
|
+
"RecordedEvidence",
|
|
39
|
+
"VariantEvidence",
|
|
40
|
+
"read_recorded_evidence",
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
#: The status a provider refuses a call with when it is being asked for too
|
|
44
|
+
#: much too quickly. Read from the recorded call rather than matched against
|
|
45
|
+
#: an error message, because a message is the provider's prose and a status is
|
|
46
|
+
#: the provider's answer.
|
|
47
|
+
RATE_LIMIT_STATUS: Final = 429
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass(frozen=True)
|
|
51
|
+
class VariantEvidence:
|
|
52
|
+
"""What one side's own recorded files say it did."""
|
|
53
|
+
|
|
54
|
+
model_turns: int
|
|
55
|
+
rollouts: int
|
|
56
|
+
rollouts_completed: int
|
|
57
|
+
rate_limited_calls: int
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@dataclass(frozen=True)
|
|
61
|
+
class RecordedEvidence:
|
|
62
|
+
"""Both sides, read from files the signed record commits by digest."""
|
|
63
|
+
|
|
64
|
+
baseline: VariantEvidence
|
|
65
|
+
candidate: VariantEvidence
|
|
66
|
+
|
|
67
|
+
@property
|
|
68
|
+
def every_rollout_completed(self) -> bool:
|
|
69
|
+
"""Whether every rollout on both sides ran to completion."""
|
|
70
|
+
return all(
|
|
71
|
+
side.rollouts_completed == side.rollouts and side.rollouts > 0
|
|
72
|
+
for side in (self.baseline, self.candidate)
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def read_recorded_evidence(
|
|
77
|
+
run_root: Path, record: ComparisonExecutionRecord
|
|
78
|
+
) -> RecordedEvidence | None:
|
|
79
|
+
"""Return both sides' counts, or ``None`` when they cannot be trusted.
|
|
80
|
+
|
|
81
|
+
``None`` is returned for every reason a reading can fail — a run whose
|
|
82
|
+
evaluation output has been cleared away, a file that no longer hashes to
|
|
83
|
+
what the record committed, a line that does not parse. There is no partial
|
|
84
|
+
answer: a result that showed one side's turns and not the other's would
|
|
85
|
+
invite exactly the comparison it could not support.
|
|
86
|
+
"""
|
|
87
|
+
paths = RunPaths(root=run_root)
|
|
88
|
+
sides = {}
|
|
89
|
+
for variant in (VariantName.BASELINE, VariantName.CANDIDATE):
|
|
90
|
+
summary = record.side(ExperimentVariant(variant.value))
|
|
91
|
+
side = _variant_evidence(
|
|
92
|
+
paths=paths,
|
|
93
|
+
variant=variant,
|
|
94
|
+
normalized_episodes_digest=summary.normalized_episodes_digest,
|
|
95
|
+
raw_traces_digest=summary.raw_traces_digest,
|
|
96
|
+
)
|
|
97
|
+
if side is None:
|
|
98
|
+
return None
|
|
99
|
+
sides[variant] = side
|
|
100
|
+
return RecordedEvidence(
|
|
101
|
+
baseline=sides[VariantName.BASELINE], candidate=sides[VariantName.CANDIDATE]
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _variant_evidence(
|
|
106
|
+
*,
|
|
107
|
+
paths: RunPaths,
|
|
108
|
+
variant: VariantName,
|
|
109
|
+
normalized_episodes_digest: str,
|
|
110
|
+
raw_traces_digest: str,
|
|
111
|
+
) -> VariantEvidence | None:
|
|
112
|
+
"""Read one side's two committed files, or return nothing."""
|
|
113
|
+
episodes_path = paths.variant_normalized_episodes(variant)
|
|
114
|
+
if _checked(episodes_path, normalized_episodes_digest) is None:
|
|
115
|
+
return None
|
|
116
|
+
try:
|
|
117
|
+
episodes = read_normalized_episodes(episodes_path)
|
|
118
|
+
except ValidationError:
|
|
119
|
+
return None
|
|
120
|
+
|
|
121
|
+
traces_path = paths.variant_output_dir(variant) / TRACES_FILENAME
|
|
122
|
+
raw = _checked(traces_path, raw_traces_digest)
|
|
123
|
+
if raw is None:
|
|
124
|
+
return None
|
|
125
|
+
rate_limited = _rate_limited_calls(raw)
|
|
126
|
+
if rate_limited is None:
|
|
127
|
+
return None
|
|
128
|
+
|
|
129
|
+
rollouts = [trace for episode in episodes for trace in episode.traces]
|
|
130
|
+
return VariantEvidence(
|
|
131
|
+
model_turns=sum(trace.num_turns for trace in rollouts),
|
|
132
|
+
rollouts=len(rollouts),
|
|
133
|
+
rollouts_completed=sum(1 for trace in rollouts if trace.ok),
|
|
134
|
+
rate_limited_calls=rate_limited,
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _checked(path: Path, digest: str) -> bytes | None:
|
|
139
|
+
"""Return a file's bytes when they are still the bytes that were signed."""
|
|
140
|
+
try:
|
|
141
|
+
data = path.read_bytes()
|
|
142
|
+
except OSError:
|
|
143
|
+
return None
|
|
144
|
+
return data if sha256_digest_bytes(data) == digest else None
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _rate_limited_calls(raw: bytes) -> int | None:
|
|
148
|
+
"""Count the model calls the provider refused with a rate limit.
|
|
149
|
+
|
|
150
|
+
The raw trace records one entry per model call, and a call the provider
|
|
151
|
+
turned away carries the status it was turned away with. Counting the
|
|
152
|
+
statuses is all this does; the message beside each one is the provider's
|
|
153
|
+
own prose about a prompt, and it is never read.
|
|
154
|
+
"""
|
|
155
|
+
total = 0
|
|
156
|
+
try:
|
|
157
|
+
for line in raw.decode("utf-8").splitlines():
|
|
158
|
+
if not line.strip():
|
|
159
|
+
continue
|
|
160
|
+
for trace in json.loads(line).get("traces") or []:
|
|
161
|
+
for call in trace.get("calls") or []:
|
|
162
|
+
error = call.get("error")
|
|
163
|
+
if isinstance(error, dict) and error.get("status_code") == (
|
|
164
|
+
RATE_LIMIT_STATUS
|
|
165
|
+
):
|
|
166
|
+
total += 1
|
|
167
|
+
except (UnicodeDecodeError, ValueError, AttributeError):
|
|
168
|
+
return None
|
|
169
|
+
return total
|