techtree 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. techtree/__init__.py +35 -0
  2. techtree/__main__.py +14 -0
  3. techtree/canonical.py +239 -0
  4. techtree/catalog/__init__.py +25 -0
  5. techtree/catalog/repository.py +400 -0
  6. techtree/catalog/service.py +419 -0
  7. techtree/cli/__init__.py +1 -0
  8. techtree/cli/app.py +416 -0
  9. techtree/cli/commands/__init__.py +1 -0
  10. techtree/cli/commands/climb.py +1223 -0
  11. techtree/cli/commands/doctor.py +147 -0
  12. techtree/cli/commands/engine.py +207 -0
  13. techtree/cli/commands/proof.py +556 -0
  14. techtree/cli/commands/publish.py +447 -0
  15. techtree/cli/commands/release.py +303 -0
  16. techtree/cli/commands/run.py +1067 -0
  17. techtree/cli/commands/setup.py +181 -0
  18. techtree/cli/commands/skill.py +221 -0
  19. techtree/cli/commands/uplift.py +698 -0
  20. techtree/cli/commands/withdraw.py +212 -0
  21. techtree/cli/confirm.py +47 -0
  22. techtree/cli/context.py +96 -0
  23. techtree/cli/invoke.py +220 -0
  24. techtree/cli/output.py +280 -0
  25. techtree/constants.py +138 -0
  26. techtree/crypto.py +128 -0
  27. techtree/doctor/__init__.py +1 -0
  28. techtree/doctor/checks.py +675 -0
  29. techtree/doctor/execution_checks.py +435 -0
  30. techtree/doctor/service.py +326 -0
  31. techtree/drafts/__init__.py +32 -0
  32. techtree/drafts/source.py +146 -0
  33. techtree/drafts/store.py +992 -0
  34. techtree/engines/__init__.py +1 -0
  35. techtree/engines/bundle.py +251 -0
  36. techtree/engines/installer.py +679 -0
  37. techtree/engines/registry.py +235 -0
  38. techtree/engines/runner.py +170 -0
  39. techtree/errors.py +262 -0
  40. techtree/fs.py +234 -0
  41. techtree/harness.py +108 -0
  42. techtree/identity/__init__.py +41 -0
  43. techtree/identity/models.py +113 -0
  44. techtree/identity/service.py +199 -0
  45. techtree/identity/store.py +263 -0
  46. techtree/ids.py +85 -0
  47. techtree/manifests/__init__.py +39 -0
  48. techtree/manifests/builder.py +433 -0
  49. techtree/manifests/compare.py +376 -0
  50. techtree/models/__init__.py +282 -0
  51. techtree/models/base.py +201 -0
  52. techtree/models/campaign.py +484 -0
  53. techtree/models/catalog.py +227 -0
  54. techtree/models/cli.py +151 -0
  55. techtree/models/climb.py +254 -0
  56. techtree/models/data_policy.py +130 -0
  57. techtree/models/engine.py +156 -0
  58. techtree/models/episode_receipt.py +130 -0
  59. techtree/models/evaluation_backend.py +113 -0
  60. techtree/models/experiment.py +154 -0
  61. techtree/models/run.py +214 -0
  62. techtree/models/skill.py +156 -0
  63. techtree/models/uplift_report.py +158 -0
  64. techtree/models/validation.py +299 -0
  65. techtree/paths.py +116 -0
  66. techtree/presentation/__init__.py +31 -0
  67. techtree/presentation/build.py +1242 -0
  68. techtree/presentation/compact.py +246 -0
  69. techtree/presentation/evidence.py +169 -0
  70. techtree/presentation/models.py +358 -0
  71. techtree/presentation/rich.py +312 -0
  72. techtree/presentation/sanitize.py +156 -0
  73. techtree/publication/__init__.py +44 -0
  74. techtree/publication/address.py +180 -0
  75. techtree/publication/coordinates.py +26 -0
  76. techtree/publication/journal.py +212 -0
  77. techtree/publication/keccak.py +183 -0
  78. techtree/publication/models.py +209 -0
  79. techtree/publication/offer.py +35 -0
  80. techtree/publication/service.py +618 -0
  81. techtree/publication/transport.py +296 -0
  82. techtree/publication/verify.py +242 -0
  83. techtree/publication/withdraw.py +156 -0
  84. techtree/py.typed +0 -0
  85. techtree/receipts/__init__.py +52 -0
  86. techtree/receipts/bundle.py +578 -0
  87. techtree/receipts/compare.py +1065 -0
  88. techtree/receipts/episode.py +672 -0
  89. techtree/receipts/execution.py +630 -0
  90. techtree/receipts/observed.py +474 -0
  91. techtree/receipts/set.py +336 -0
  92. techtree/receipts/uplift.py +655 -0
  93. techtree/receipts/verify.py +1055 -0
  94. techtree/release/__init__.py +9 -0
  95. techtree/release/bootstrap.py +509 -0
  96. techtree/release/checks.py +376 -0
  97. techtree/release/document.py +125 -0
  98. techtree/release/generate.py +221 -0
  99. techtree/release/models.py +293 -0
  100. techtree/release/provenance.py +109 -0
  101. techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
  102. techtree/resources/catalog/catalog.json +32 -0
  103. techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
  104. techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
  105. techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
  106. techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
  107. techtree/resources/engines/default/engine.json +20 -0
  108. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
  109. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
  110. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
  111. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
  112. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
  113. techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
  114. techtree/resources/engines/default/pyproject.toml +23 -0
  115. techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
  116. techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
  117. techtree/resources/engines/default/tools/normalize_validation.py +222 -0
  118. techtree/resources/engines/default/uv.lock +1758 -0
  119. techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
  120. techtree/resources/release/build-provenance.json +4 -0
  121. techtree/resources/release/release-core.json +24 -0
  122. techtree/runs/__init__.py +31 -0
  123. techtree/runs/artifacts.py +750 -0
  124. techtree/runs/child_registry.py +228 -0
  125. techtree/runs/events.py +478 -0
  126. techtree/runs/executor.py +140 -0
  127. techtree/runs/fake.py +741 -0
  128. techtree/runs/launcher.py +253 -0
  129. techtree/runs/machine.py +489 -0
  130. techtree/runs/real.py +789 -0
  131. techtree/runs/service.py +616 -0
  132. techtree/runs/store.py +555 -0
  133. techtree/runs/validation.py +259 -0
  134. techtree/runs/variants.py +684 -0
  135. techtree/settings.py +143 -0
  136. techtree/skills/__init__.py +14 -0
  137. techtree/skills/archive.py +282 -0
  138. techtree/skills/policy.py +62 -0
  139. techtree/skills/scanner.py +394 -0
  140. techtree/skills/service.py +752 -0
  141. techtree/skills/starter.py +434 -0
  142. techtree/tasksets/__init__.py +1 -0
  143. techtree/tasksets/membership.py +269 -0
  144. techtree/tasksets/provider.py +207 -0
  145. techtree/tasksets/resolver.py +311 -0
  146. techtree/tasksets/service.py +484 -0
  147. techtree/tasksets/verifiers_cli.py +538 -0
  148. techtree/uplift/__init__.py +20 -0
  149. techtree/uplift/context.py +544 -0
  150. techtree/uplift/derive.py +203 -0
  151. techtree/uplift/public_tasks.py +151 -0
  152. techtree/uplift/service.py +719 -0
  153. techtree/uplift/source.py +160 -0
  154. techtree/verifiers/__init__.py +31 -0
  155. techtree/verifiers/budget.py +219 -0
  156. techtree/verifiers/child.py +633 -0
  157. techtree/verifiers/compiler.py +432 -0
  158. techtree/verifiers/config.py +365 -0
  159. techtree/verifiers/credentials.py +321 -0
  160. techtree/verifiers/image.py +126 -0
  161. techtree/verifiers/models.py +527 -0
  162. techtree/verifiers/outputs.py +368 -0
  163. techtree/verifiers/progress.py +192 -0
  164. techtree/verifiers/supervisor.py +341 -0
  165. techtree/verifiers/verify.py +782 -0
  166. techtree/version.py +39 -0
  167. techtree/worker/__init__.py +18 -0
  168. techtree/worker/execute.py +487 -0
  169. techtree/worker/main.py +57 -0
  170. techtree-0.1.0.dist-info/METADATA +344 -0
  171. techtree-0.1.0.dist-info/RECORD +174 -0
  172. techtree-0.1.0.dist-info/WHEEL +4 -0
  173. techtree-0.1.0.dist-info/entry_points.txt +3 -0
  174. techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,246 @@
1
+ """The rendering a phone can carry. Spec section 7.16.
2
+
3
+ A gateway message is not a small terminal. It has no escape sequences, no
4
+ column alignment, a reader holding a phone, and a channel that may truncate
5
+ anything long. So this renderer is bounded by construction: a headline, the
6
+ counts, the honest qualifications, at most a few task rows, and one sentence
7
+ about what could happen next.
8
+
9
+ The row cap is a default, not a refusal. A reader who says ``--show-tasks all``
10
+ has asked for every task on purpose, and the bound exists so that an unasked-for
11
+ message is short rather than so that somebody who asked can be told no. So the
12
+ cap applies to the selections a reader did not have to name, and the explicit
13
+ request for everything overrides it.
14
+
15
+ Two things are *not* dropped to make it fit.
16
+
17
+ *Every warning and error survives.* Room is made by cutting the table, never by
18
+ cutting a caveat. A result that only says what went well on a phone and keeps
19
+ its qualifications for the desktop would be dishonest in exactly the channel
20
+ where somebody is most likely to forward it to someone else.
21
+
22
+ *The proof grade travels with the numbers.* A one-line quote of an uplift with
23
+ no grade beside it is the shape misinformation takes, so the grade and whether
24
+ its proof verified are part of the same block as the scores.
25
+
26
+ Nothing here emits ANSI, and nothing here formats a number differently from the
27
+ terminal renderer: both read the same payload fields.
28
+ """
29
+
30
+ from __future__ import annotations
31
+
32
+ from typing import Final
33
+
34
+ from techtree.presentation.build import (
35
+ HELD_FIXED_LINE,
36
+ NOT_BROAD_CAPABILITY_LINE,
37
+ VERIFICATION_FAILED,
38
+ VERIFICATION_NOT_VERIFIED,
39
+ VERIFICATION_VERIFIED,
40
+ cost_explanation,
41
+ cost_summary,
42
+ decision_headline,
43
+ efficiency_sentence,
44
+ solved_line,
45
+ task_count_line,
46
+ )
47
+ from techtree.presentation.models import (
48
+ TaskDisplay,
49
+ UpliftPresentationPayload,
50
+ selected_task_rows,
51
+ )
52
+
53
+ __all__ = [
54
+ "DEFAULT_MAXIMUM_TASK_ROWS",
55
+ "UNVERIFIED_HEADLINE",
56
+ "render_uplift_markdown",
57
+ ]
58
+
59
+ #: Enough rows to show a pattern, few enough to read on a phone.
60
+ DEFAULT_MAXIMUM_TASK_ROWS: Final = 5
61
+
62
+ #: What a result whose proof failed says before it says anything else.
63
+ UNVERIFIED_HEADLINE: Final = (
64
+ "**This run's local proof did not verify. Do not rely on the numbers below.**"
65
+ )
66
+
67
+ _VERIFICATION_PHRASE: Final[dict[str, str]] = {
68
+ VERIFICATION_VERIFIED: "signature verified offline",
69
+ VERIFICATION_FAILED: "signature DID NOT verify",
70
+ VERIFICATION_NOT_VERIFIED: "signature not checked",
71
+ }
72
+
73
+
74
+ def render_uplift_markdown(
75
+ payload: UpliftPresentationPayload,
76
+ *,
77
+ maximum_task_rows: int = DEFAULT_MAXIMUM_TASK_ROWS,
78
+ show_tasks: TaskDisplay = TaskDisplay.CHANGED,
79
+ ) -> str:
80
+ """Return compact Markdown suitable for a phone or gateway message.
81
+
82
+ A result whose proof did not verify says that first and in bold. This is
83
+ the channel a number is most likely to be quoted out of, so the sentence
84
+ that would stop somebody quoting it cannot be further down.
85
+
86
+ ``show_tasks`` is the same choice the terminal renderer takes, answered by
87
+ the same reader through the same option, and it selects rows through the
88
+ same rule. A reader who asks a piped command for every task and is handed
89
+ the default five has been told the option works when it did not.
90
+ """
91
+ lines = [
92
+ f"**{decision_headline(payload)} — {solved_line(payload)}**",
93
+ "",
94
+ f"- {NOT_BROAD_CAPABILITY_LINE}",
95
+ ]
96
+ if payload.verification_status == VERIFICATION_FAILED:
97
+ lines = [UNVERIFIED_HEADLINE, "", *lines]
98
+ lines += [
99
+ f"- {payload.campaign_title} — {payload.comparison_label}",
100
+ f"- Changed: {payload.change_label}. {HELD_FIXED_LINE}",
101
+ f"- Tasks: {_headline_numbers(payload)}",
102
+ f"- Proof: local {payload.proof_grade}, "
103
+ f"{_VERIFICATION_PHRASE[payload.verification_status]}",
104
+ f"- Cost: {cost_summary(payload)}",
105
+ *(f"- {line}" for line in cost_explanation(payload)),
106
+ *_work(payload),
107
+ "- Raw episodes: retained locally; not uploaded",
108
+ ]
109
+
110
+ lines += _table(payload, show_tasks, maximum_task_rows)
111
+
112
+ qualifications = [caveat for caveat in payload.caveats if caveat.severity != "info"]
113
+ if qualifications:
114
+ lines.append("")
115
+ lines.extend(f"- {caveat.text}" for caveat in qualifications)
116
+
117
+ lines.append("")
118
+ lines.append(_next_line(payload))
119
+ return "\n".join(lines)
120
+
121
+
122
+ def _headline_numbers(payload: UpliftPresentationPayload) -> str:
123
+ """Return how the two sides scored, in the unit a person counts in.
124
+
125
+ The bold line above carries what was established and what is still failing.
126
+ This carries the movement underneath it: both sides' counts where the
127
+ reward has them, and the means they came from in the same breath, so that
128
+ nothing is lost by quoting either one.
129
+ """
130
+ means = (
131
+ f"{payload.baseline_score:.3f} → {payload.candidate_score:.3f} "
132
+ f"({payload.absolute_delta:+.3f})"
133
+ )
134
+ counted = task_count_line(payload)
135
+ if counted is None:
136
+ return f"mean {means}"
137
+ return f"{counted}, mean {means}"
138
+
139
+
140
+ def _work(payload: UpliftPresentationPayload) -> list[str]:
141
+ """Return what each side spent doing the same tasks, in one bullet.
142
+
143
+ The channel with the least room still carries this: a Skill that took a
144
+ third of the model turns did something a reader wants to know, and unlike
145
+ the clock, that number does not move when the same run is repeated on a
146
+ busier afternoon. The sentence carries both times, so a run that has it
147
+ does not also get a bare pair of durations.
148
+ """
149
+ sentence = efficiency_sentence(payload)
150
+ if sentence is not None:
151
+ return [f"- Work: {sentence}"]
152
+ return [f"- Time: {_time(payload)}"]
153
+
154
+
155
+ def _time(payload: UpliftPresentationPayload) -> str:
156
+ """Return how long each side took, or say it was not recorded.
157
+
158
+ Decisions document 0019 section 3 puts timing in the measured difference,
159
+ so the channel with the least room still carries it: a comparison whose
160
+ two sides took very different amounts of time is a comparison a reader
161
+ should be able to ask about.
162
+ """
163
+ baseline = payload.baseline_seconds
164
+ candidate = payload.candidate_seconds
165
+ if baseline is None and candidate is None:
166
+ return "not recorded for this run"
167
+ return f"baseline {_seconds(baseline)}, candidate {_seconds(candidate)}"
168
+
169
+
170
+ def _seconds(value: float | None) -> str:
171
+ """Return one side's elapsed time, or the word for an absent one."""
172
+ if value is None:
173
+ return "unavailable"
174
+ return f"{value:.1f}s"
175
+
176
+
177
+ def _next_line(payload: UpliftPresentationPayload) -> str:
178
+ """Offer, in one line, the steps this result actually has.
179
+
180
+ A development-only result has no proof to check and nothing may be derived
181
+ from it, so it is offered the one thing it can do. Everything else is left
182
+ unsaid rather than promised on a channel with no room to explain.
183
+ """
184
+ if payload.proof_grade == "development_only":
185
+ return "Next: I can show every task locally."
186
+ return (
187
+ "Next: I can show every task locally, set up a comparison against a "
188
+ "revised Skill, or check this run's local receipt offline with "
189
+ "`techtree proof verify`."
190
+ )
191
+
192
+
193
+ def _table(
194
+ payload: UpliftPresentationPayload, show: TaskDisplay, maximum_task_rows: int
195
+ ) -> list[str]:
196
+ """Return the heading and the rows for the table this reader asked for.
197
+
198
+ The cap is the bound this channel keeps by default, and it is applied to
199
+ every selection except the one a reader had to type out. Asking for all
200
+ tasks is a deliberate override of a default, so it is honoured; a reader
201
+ who wanted a short message never asked for the long one.
202
+
203
+ A reader who asked for no table, or a selection with nothing in it, gets
204
+ neither rows nor a heading over them. A heading with nothing underneath is
205
+ a line that says a table exists somewhere it does not.
206
+ """
207
+ selected = selected_task_rows(payload.task_rows, show)
208
+ shown = selected if show is TaskDisplay.ALL else selected[:maximum_task_rows]
209
+ if not shown:
210
+ return []
211
+ return [
212
+ "",
213
+ _heading(payload, show, shown=len(shown), selected=len(selected)),
214
+ *(
215
+ f"- {row.task_label}: {row.baseline_score:.2f} → "
216
+ f"{row.candidate_score:.2f} ({row.outcome.upper()})"
217
+ for row in shown
218
+ ),
219
+ ]
220
+
221
+
222
+ def _heading(
223
+ payload: UpliftPresentationPayload,
224
+ show: TaskDisplay,
225
+ *,
226
+ shown: int,
227
+ selected: int,
228
+ ) -> str:
229
+ """Say what the rows underneath are, and say when some were left out.
230
+
231
+ The heading a reader can check is the heading that names its own
232
+ selection. Calling a full table the largest changes claims a ranking that
233
+ nothing performed, and calling a list of losses by the same name reads as
234
+ though the wins had also been in the running and only just lost.
235
+
236
+ The counts follow the same rule. Where every row a reader asked for is
237
+ printed, the pair says how much of the whole comparison that is; where the
238
+ cap left some out, it says how many of the asked-for rows are here instead,
239
+ because that is the number that tells a reader something is missing.
240
+ """
241
+ if show is TaskDisplay.ALL:
242
+ return f"All {selected} tasks:"
243
+ name = "Regressions" if show is TaskDisplay.REGRESSIONS else "Changed tasks"
244
+ if shown < selected:
245
+ return f"{name} ({shown} of {selected} shown):"
246
+ return f"{name} ({selected} of {len(payload.task_rows)} tasks):"
@@ -0,0 +1,169 @@
1
+ """Counts read back out of a finished run's own recorded evidence.
2
+
3
+ Two facts a reader of a result asks for are recorded by every run and carried
4
+ by none of its signed documents: how many model turns each side took, and how
5
+ often the provider refused a model call. The first is the efficiency finding a
6
+ Skill's whole value can sit in; the second is a validity question, because two
7
+ sides that met different amounts of throttling did not meet identical
8
+ conditions even when both finished.
9
+
10
+ Neither is added to a signed artifact here. The signed
11
+ :class:`~techtree.receipts.execution.ComparisonExecutionRecord` already commits
12
+ each side's normalized episodes and raw traces *by digest*, so this reads those
13
+ two files back at render time and checks them against the digests the record
14
+ already holds. A file that is missing, unreadable, or no longer the file the
15
+ record committed yields nothing at all rather than a number: an unchecked count
16
+ would be worth less than saying it is unknown.
17
+
18
+ Only counts leave this module. The files it opens carry prompts, replies and
19
+ grader material, and nothing here returns a string taken from either of them.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import json
25
+ from dataclasses import dataclass
26
+ from pathlib import Path
27
+ from typing import Final
28
+
29
+ from techtree.canonical import sha256_digest_bytes
30
+ from techtree.errors import ValidationError
31
+ from techtree.models.experiment import ExperimentVariant
32
+ from techtree.receipts.execution import ComparisonExecutionRecord
33
+ from techtree.verifiers.models import RunPaths, VariantName
34
+ from techtree.verifiers.outputs import TRACES_FILENAME, read_normalized_episodes
35
+
36
+ __all__ = [
37
+ "RATE_LIMIT_STATUS",
38
+ "RecordedEvidence",
39
+ "VariantEvidence",
40
+ "read_recorded_evidence",
41
+ ]
42
+
43
+ #: The status a provider refuses a call with when it is being asked for too
44
+ #: much too quickly. Read from the recorded call rather than matched against
45
+ #: an error message, because a message is the provider's prose and a status is
46
+ #: the provider's answer.
47
+ RATE_LIMIT_STATUS: Final = 429
48
+
49
+
50
+ @dataclass(frozen=True)
51
+ class VariantEvidence:
52
+ """What one side's own recorded files say it did."""
53
+
54
+ model_turns: int
55
+ rollouts: int
56
+ rollouts_completed: int
57
+ rate_limited_calls: int
58
+
59
+
60
+ @dataclass(frozen=True)
61
+ class RecordedEvidence:
62
+ """Both sides, read from files the signed record commits by digest."""
63
+
64
+ baseline: VariantEvidence
65
+ candidate: VariantEvidence
66
+
67
+ @property
68
+ def every_rollout_completed(self) -> bool:
69
+ """Whether every rollout on both sides ran to completion."""
70
+ return all(
71
+ side.rollouts_completed == side.rollouts and side.rollouts > 0
72
+ for side in (self.baseline, self.candidate)
73
+ )
74
+
75
+
76
+ def read_recorded_evidence(
77
+ run_root: Path, record: ComparisonExecutionRecord
78
+ ) -> RecordedEvidence | None:
79
+ """Return both sides' counts, or ``None`` when they cannot be trusted.
80
+
81
+ ``None`` is returned for every reason a reading can fail — a run whose
82
+ evaluation output has been cleared away, a file that no longer hashes to
83
+ what the record committed, a line that does not parse. There is no partial
84
+ answer: a result that showed one side's turns and not the other's would
85
+ invite exactly the comparison it could not support.
86
+ """
87
+ paths = RunPaths(root=run_root)
88
+ sides = {}
89
+ for variant in (VariantName.BASELINE, VariantName.CANDIDATE):
90
+ summary = record.side(ExperimentVariant(variant.value))
91
+ side = _variant_evidence(
92
+ paths=paths,
93
+ variant=variant,
94
+ normalized_episodes_digest=summary.normalized_episodes_digest,
95
+ raw_traces_digest=summary.raw_traces_digest,
96
+ )
97
+ if side is None:
98
+ return None
99
+ sides[variant] = side
100
+ return RecordedEvidence(
101
+ baseline=sides[VariantName.BASELINE], candidate=sides[VariantName.CANDIDATE]
102
+ )
103
+
104
+
105
+ def _variant_evidence(
106
+ *,
107
+ paths: RunPaths,
108
+ variant: VariantName,
109
+ normalized_episodes_digest: str,
110
+ raw_traces_digest: str,
111
+ ) -> VariantEvidence | None:
112
+ """Read one side's two committed files, or return nothing."""
113
+ episodes_path = paths.variant_normalized_episodes(variant)
114
+ if _checked(episodes_path, normalized_episodes_digest) is None:
115
+ return None
116
+ try:
117
+ episodes = read_normalized_episodes(episodes_path)
118
+ except ValidationError:
119
+ return None
120
+
121
+ traces_path = paths.variant_output_dir(variant) / TRACES_FILENAME
122
+ raw = _checked(traces_path, raw_traces_digest)
123
+ if raw is None:
124
+ return None
125
+ rate_limited = _rate_limited_calls(raw)
126
+ if rate_limited is None:
127
+ return None
128
+
129
+ rollouts = [trace for episode in episodes for trace in episode.traces]
130
+ return VariantEvidence(
131
+ model_turns=sum(trace.num_turns for trace in rollouts),
132
+ rollouts=len(rollouts),
133
+ rollouts_completed=sum(1 for trace in rollouts if trace.ok),
134
+ rate_limited_calls=rate_limited,
135
+ )
136
+
137
+
138
+ def _checked(path: Path, digest: str) -> bytes | None:
139
+ """Return a file's bytes when they are still the bytes that were signed."""
140
+ try:
141
+ data = path.read_bytes()
142
+ except OSError:
143
+ return None
144
+ return data if sha256_digest_bytes(data) == digest else None
145
+
146
+
147
+ def _rate_limited_calls(raw: bytes) -> int | None:
148
+ """Count the model calls the provider refused with a rate limit.
149
+
150
+ The raw trace records one entry per model call, and a call the provider
151
+ turned away carries the status it was turned away with. Counting the
152
+ statuses is all this does; the message beside each one is the provider's
153
+ own prose about a prompt, and it is never read.
154
+ """
155
+ total = 0
156
+ try:
157
+ for line in raw.decode("utf-8").splitlines():
158
+ if not line.strip():
159
+ continue
160
+ for trace in json.loads(line).get("traces") or []:
161
+ for call in trace.get("calls") or []:
162
+ error = call.get("error")
163
+ if isinstance(error, dict) and error.get("status_code") == (
164
+ RATE_LIMIT_STATUS
165
+ ):
166
+ total += 1
167
+ except (UnicodeDecodeError, ValueError, AttributeError):
168
+ return None
169
+ return total