techtree 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. techtree/__init__.py +35 -0
  2. techtree/__main__.py +14 -0
  3. techtree/canonical.py +239 -0
  4. techtree/catalog/__init__.py +25 -0
  5. techtree/catalog/repository.py +400 -0
  6. techtree/catalog/service.py +419 -0
  7. techtree/cli/__init__.py +1 -0
  8. techtree/cli/app.py +416 -0
  9. techtree/cli/commands/__init__.py +1 -0
  10. techtree/cli/commands/climb.py +1223 -0
  11. techtree/cli/commands/doctor.py +147 -0
  12. techtree/cli/commands/engine.py +207 -0
  13. techtree/cli/commands/proof.py +556 -0
  14. techtree/cli/commands/publish.py +447 -0
  15. techtree/cli/commands/release.py +303 -0
  16. techtree/cli/commands/run.py +1067 -0
  17. techtree/cli/commands/setup.py +181 -0
  18. techtree/cli/commands/skill.py +221 -0
  19. techtree/cli/commands/uplift.py +698 -0
  20. techtree/cli/commands/withdraw.py +212 -0
  21. techtree/cli/confirm.py +47 -0
  22. techtree/cli/context.py +96 -0
  23. techtree/cli/invoke.py +220 -0
  24. techtree/cli/output.py +280 -0
  25. techtree/constants.py +138 -0
  26. techtree/crypto.py +128 -0
  27. techtree/doctor/__init__.py +1 -0
  28. techtree/doctor/checks.py +675 -0
  29. techtree/doctor/execution_checks.py +435 -0
  30. techtree/doctor/service.py +326 -0
  31. techtree/drafts/__init__.py +32 -0
  32. techtree/drafts/source.py +146 -0
  33. techtree/drafts/store.py +992 -0
  34. techtree/engines/__init__.py +1 -0
  35. techtree/engines/bundle.py +251 -0
  36. techtree/engines/installer.py +679 -0
  37. techtree/engines/registry.py +235 -0
  38. techtree/engines/runner.py +170 -0
  39. techtree/errors.py +262 -0
  40. techtree/fs.py +234 -0
  41. techtree/harness.py +108 -0
  42. techtree/identity/__init__.py +41 -0
  43. techtree/identity/models.py +113 -0
  44. techtree/identity/service.py +199 -0
  45. techtree/identity/store.py +263 -0
  46. techtree/ids.py +85 -0
  47. techtree/manifests/__init__.py +39 -0
  48. techtree/manifests/builder.py +433 -0
  49. techtree/manifests/compare.py +376 -0
  50. techtree/models/__init__.py +282 -0
  51. techtree/models/base.py +201 -0
  52. techtree/models/campaign.py +484 -0
  53. techtree/models/catalog.py +227 -0
  54. techtree/models/cli.py +151 -0
  55. techtree/models/climb.py +254 -0
  56. techtree/models/data_policy.py +130 -0
  57. techtree/models/engine.py +156 -0
  58. techtree/models/episode_receipt.py +130 -0
  59. techtree/models/evaluation_backend.py +113 -0
  60. techtree/models/experiment.py +154 -0
  61. techtree/models/run.py +214 -0
  62. techtree/models/skill.py +156 -0
  63. techtree/models/uplift_report.py +158 -0
  64. techtree/models/validation.py +299 -0
  65. techtree/paths.py +116 -0
  66. techtree/presentation/__init__.py +31 -0
  67. techtree/presentation/build.py +1242 -0
  68. techtree/presentation/compact.py +246 -0
  69. techtree/presentation/evidence.py +169 -0
  70. techtree/presentation/models.py +358 -0
  71. techtree/presentation/rich.py +312 -0
  72. techtree/presentation/sanitize.py +156 -0
  73. techtree/publication/__init__.py +44 -0
  74. techtree/publication/address.py +180 -0
  75. techtree/publication/coordinates.py +26 -0
  76. techtree/publication/journal.py +212 -0
  77. techtree/publication/keccak.py +183 -0
  78. techtree/publication/models.py +209 -0
  79. techtree/publication/offer.py +35 -0
  80. techtree/publication/service.py +618 -0
  81. techtree/publication/transport.py +296 -0
  82. techtree/publication/verify.py +242 -0
  83. techtree/publication/withdraw.py +156 -0
  84. techtree/py.typed +0 -0
  85. techtree/receipts/__init__.py +52 -0
  86. techtree/receipts/bundle.py +578 -0
  87. techtree/receipts/compare.py +1065 -0
  88. techtree/receipts/episode.py +672 -0
  89. techtree/receipts/execution.py +630 -0
  90. techtree/receipts/observed.py +474 -0
  91. techtree/receipts/set.py +336 -0
  92. techtree/receipts/uplift.py +655 -0
  93. techtree/receipts/verify.py +1055 -0
  94. techtree/release/__init__.py +9 -0
  95. techtree/release/bootstrap.py +509 -0
  96. techtree/release/checks.py +376 -0
  97. techtree/release/document.py +125 -0
  98. techtree/release/generate.py +221 -0
  99. techtree/release/models.py +293 -0
  100. techtree/release/provenance.py +109 -0
  101. techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
  102. techtree/resources/catalog/catalog.json +32 -0
  103. techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
  104. techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
  105. techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
  106. techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
  107. techtree/resources/engines/default/engine.json +20 -0
  108. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
  109. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
  110. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
  111. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
  112. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
  113. techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
  114. techtree/resources/engines/default/pyproject.toml +23 -0
  115. techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
  116. techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
  117. techtree/resources/engines/default/tools/normalize_validation.py +222 -0
  118. techtree/resources/engines/default/uv.lock +1758 -0
  119. techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
  120. techtree/resources/release/build-provenance.json +4 -0
  121. techtree/resources/release/release-core.json +24 -0
  122. techtree/runs/__init__.py +31 -0
  123. techtree/runs/artifacts.py +750 -0
  124. techtree/runs/child_registry.py +228 -0
  125. techtree/runs/events.py +478 -0
  126. techtree/runs/executor.py +140 -0
  127. techtree/runs/fake.py +741 -0
  128. techtree/runs/launcher.py +253 -0
  129. techtree/runs/machine.py +489 -0
  130. techtree/runs/real.py +789 -0
  131. techtree/runs/service.py +616 -0
  132. techtree/runs/store.py +555 -0
  133. techtree/runs/validation.py +259 -0
  134. techtree/runs/variants.py +684 -0
  135. techtree/settings.py +143 -0
  136. techtree/skills/__init__.py +14 -0
  137. techtree/skills/archive.py +282 -0
  138. techtree/skills/policy.py +62 -0
  139. techtree/skills/scanner.py +394 -0
  140. techtree/skills/service.py +752 -0
  141. techtree/skills/starter.py +434 -0
  142. techtree/tasksets/__init__.py +1 -0
  143. techtree/tasksets/membership.py +269 -0
  144. techtree/tasksets/provider.py +207 -0
  145. techtree/tasksets/resolver.py +311 -0
  146. techtree/tasksets/service.py +484 -0
  147. techtree/tasksets/verifiers_cli.py +538 -0
  148. techtree/uplift/__init__.py +20 -0
  149. techtree/uplift/context.py +544 -0
  150. techtree/uplift/derive.py +203 -0
  151. techtree/uplift/public_tasks.py +151 -0
  152. techtree/uplift/service.py +719 -0
  153. techtree/uplift/source.py +160 -0
  154. techtree/verifiers/__init__.py +31 -0
  155. techtree/verifiers/budget.py +219 -0
  156. techtree/verifiers/child.py +633 -0
  157. techtree/verifiers/compiler.py +432 -0
  158. techtree/verifiers/config.py +365 -0
  159. techtree/verifiers/credentials.py +321 -0
  160. techtree/verifiers/image.py +126 -0
  161. techtree/verifiers/models.py +527 -0
  162. techtree/verifiers/outputs.py +368 -0
  163. techtree/verifiers/progress.py +192 -0
  164. techtree/verifiers/supervisor.py +341 -0
  165. techtree/verifiers/verify.py +782 -0
  166. techtree/version.py +39 -0
  167. techtree/worker/__init__.py +18 -0
  168. techtree/worker/execute.py +487 -0
  169. techtree/worker/main.py +57 -0
  170. techtree-0.1.0.dist-info/METADATA +344 -0
  171. techtree-0.1.0.dist-info/RECORD +174 -0
  172. techtree-0.1.0.dist-info/WHEEL +4 -0
  173. techtree-0.1.0.dist-info/entry_points.txt +3 -0
  174. techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,544 @@
1
+ """What a host agent may be told about a finished run. Spec section 7.18.
2
+
3
+ The improvement loop needs a model to read one run's outcome and propose a
4
+ better Skill. This module is the boundary that model reads through, and its
5
+ whole design is a subtraction: it starts from what the run proved and removes
6
+ everything a Skill must not be allowed to learn.
7
+
8
+ What comes out is a :class:`SkillImprovementContext` — the objective, the
9
+ headline result, and a bounded, ordered list of tasks worth looking at. What
10
+ never comes out is listed in ``prohibited_material``, on the artifact itself,
11
+ so a consumer is told what it is not being given rather than left to guess.
12
+
13
+ Three rules decide the contents.
14
+
15
+ *It is built from the run's signed record.* The report and the episode receipts
16
+ are the objects the run's local proof covers; the engine's raw and normalized
17
+ evaluation output is not. Building from the signed half is what makes the
18
+ context deterministic — the same run always produces the same bytes — and what
19
+ stops a context from asserting something the run's own attestation does not.
20
+
21
+ *Subject replies are excluded, and that is the load-bearing decision.* Spec
22
+ section 11.5 lists subject replies among what a sanitized context may include;
23
+ spec section 7.18 requires expected answers to be excluded, and spec section
24
+ 7.22 requires a test proving a hidden answer is omitted. For a taskset scored
25
+ by matching an answer, a reply the subject got *right* is the expected answer,
26
+ word for word. Handing those to the model that writes Skill v2 would let the
27
+ answers be written into the Skill, and the v1-against-v2 comparison would then
28
+ measure memorization while presenting itself as measuring procedure — the
29
+ contaminated benchmark spec section 7.17 exists to prevent. The exclusions
30
+ control, so replies stay out. :class:`ImprovementExample` keeps the field,
31
+ because it is the typed seat a later release can fill once something in the
32
+ protocol can say which rewards make a reply safe to show; in this build it is
33
+ always ``None`` and ``prohibited_material`` says so.
34
+
35
+ *Nothing local leaks.* No filesystem path, no environment value, and no
36
+ provider detail has a field to enter through, and every free-text field is
37
+ checked for control sequences and absolute paths the same way a rendering is.
38
+ A task is named by its position and its committed hash, which is the
39
+ identifier the TasksetLock already commits to in public.
40
+
41
+ One thing is not checked, by decision 0036: nothing here looks at a string and
42
+ decides it is credential-shaped. An error summary carries whatever the engine
43
+ printed, and ``prohibited_material`` no longer claims otherwise.
44
+
45
+ The context is not signed, is not proof, and is not uploaded.
46
+ """
47
+
48
+ from __future__ import annotations
49
+
50
+ from collections.abc import Sequence
51
+ from typing import Final, Literal, Protocol
52
+
53
+ from techtree.canonical import digest_object
54
+ from techtree.errors import ValidationError
55
+ from techtree.models.base import Digest, NonEmptyString, ProtocolModel
56
+ from techtree.models.campaign import CampaignSpec
57
+ from techtree.models.episode_receipt import EpisodeReceipt, ScoreStatus
58
+ from techtree.models.skill import SKILL_ENTRY_FILE, SkillArtifact
59
+ from techtree.models.uplift_report import (
60
+ PrimaryUpliftResult,
61
+ TaskDelta,
62
+ UpliftReport,
63
+ )
64
+ from techtree.presentation.sanitize import (
65
+ ensure_no_control_or_local_path,
66
+ sanitize_label,
67
+ )
68
+
69
+ __all__ = [
70
+ "EXAMPLE_CONTRAST_LIMIT",
71
+ "EXAMPLE_LIMIT",
72
+ "IMPROVEMENT_CONTEXT_FORBIDDEN_MATERIAL",
73
+ "IMPROVEMENT_CONTEXT_INVALID",
74
+ "IMPROVEMENT_CONTEXT_SCHEMA_VERSION",
75
+ "PROHIBITED_MATERIAL",
76
+ "REVISION_CONSTRAINTS",
77
+ "ImprovementExample",
78
+ "ImprovementOutcome",
79
+ "SkillImprovementContext",
80
+ "TaskPublicProjection",
81
+ "TaskPublicProjectionProvider",
82
+ "build_improvement_context",
83
+ "entrypoint_digest",
84
+ "hash_only_projection",
85
+ ]
86
+
87
+ #: Not in :mod:`techtree.constants`, which holds protocol schema versions. Spec
88
+ #: section 7.18's context is local machine-readable material, not a protocol
89
+ #: object: nothing signs it, nothing references it by digest, and no schema is
90
+ #: exported for it.
91
+ IMPROVEMENT_CONTEXT_SCHEMA_VERSION: Final = "techtree.skill-improvement-context.v1"
92
+
93
+ #: Stable error codes. Both are spec ``docs/spec/climb-v0.1-wp9-wp11.md``
94
+ #: section 20's, named there because that is where the context is consumed and
95
+ #: used here because this is where the two conditions occur: a context that
96
+ #: could not be built honestly, and one whose free text carried something the
97
+ #: exclusion list forbids.
98
+ IMPROVEMENT_CONTEXT_INVALID: Final = "improvement_context_invalid"
99
+ IMPROVEMENT_CONTEXT_FORBIDDEN_MATERIAL: Final = "improvement_context_forbidden_material"
100
+
101
+ #: How many tasks a context carries at most. A context is read by a model with
102
+ #: a finite window, and a run with hundreds of tasks would otherwise push the
103
+ #: regressions — the part that matters most — out of reach.
104
+ EXAMPLE_LIMIT: Final = 20
105
+
106
+ #: How many stable successes ride along as contrast. Spec section 7.18 asks for
107
+ #: "a small contrast sample", and small is what keeps the list about failures.
108
+ EXAMPLE_CONTRAST_LIMIT: Final = 3
109
+
110
+
111
+ type ImprovementOutcome = Literal[
112
+ "stable_success",
113
+ "stable_failure",
114
+ "improved",
115
+ "regressed",
116
+ ]
117
+ """How one task moved between the two variants of the source run."""
118
+
119
+
120
+ #: What a revision is allowed to be. Stated on the artifact because the model
121
+ #: reading it is being asked to produce a Skill, and a constraint it was never
122
+ #: told about is a constraint it will break.
123
+ REVISION_CONSTRAINTS: Final[tuple[str, ...]] = (
124
+ "The revision is an instruction Skill: Markdown and plain text files, with "
125
+ "SKILL.md as its entry point.",
126
+ "The revision changes only the Skill's own files. Nothing else about the "
127
+ "experiment may differ, and nothing else will be allowed to.",
128
+ "The revision must not encode answers to specific tasks. It is measured on "
129
+ "the same committed tasks, and a Skill that memorizes them measures nothing.",
130
+ "The revision must not contain credentials, keys, tokens, or absolute local "
131
+ "paths. Nothing checks for them, so writing one in puts it in the record.",
132
+ "The revision must differ from the Skill it replaces. An identical tree "
133
+ "would compare a Skill against itself.",
134
+ )
135
+
136
+ #: What this context does not carry, stated to whoever reads it. Every entry is
137
+ #: proved absent by the exclusion matrix in ``tests/unit/test_improvement_context``.
138
+ PROHIBITED_MATERIAL: Final[tuple[str, ...]] = (
139
+ "hidden expected answers",
140
+ "hidden grader material",
141
+ "sealed task content",
142
+ "subject final replies",
143
+ "private environment values",
144
+ "unredacted local filesystem paths",
145
+ )
146
+
147
+
148
+ class TaskPublicProjection(ProtocolModel):
149
+ """The publicly showable face of one committed task."""
150
+
151
+ task_label: NonEmptyString
152
+ public_prompt: str | None
153
+
154
+
155
+ class TaskPublicProjectionProvider(Protocol):
156
+ """How a caller supplies the public face of one task.
157
+
158
+ The taskset's own public material is not carried in a receipt, so this
159
+ build has nothing to resolve a prompt from and
160
+ :func:`hash_only_projection` names a task by its position and committed
161
+ hash. The seat exists so that a later release with a DataPolicy-gated
162
+ public projection can fill it without changing this module.
163
+ """
164
+
165
+ def __call__(self, *, task_hash: Digest, position: int) -> TaskPublicProjection:
166
+ """Return what may be shown about one committed task."""
167
+ ...
168
+
169
+
170
+ def hash_only_projection(*, task_hash: Digest, position: int) -> TaskPublicProjection:
171
+ """Name a task by its place in the Campaign and the head of its hash.
172
+
173
+ A committed task hash is not hidden material: it is the identifier the
174
+ TasksetLock, the receipts and the report already carry, and the publisher's
175
+ validation evidence lists it in public. The prompt is absent because no
176
+ public projection of one exists here, not because one was dropped.
177
+ """
178
+ _, _, hexadecimal = task_hash.partition(":")
179
+ return TaskPublicProjection(
180
+ task_label=f"task {position + 1:02d} · {hexadecimal[:8]}",
181
+ public_prompt=None,
182
+ )
183
+
184
+
185
+ class ImprovementExample(ProtocolModel):
186
+ """One committed task, as a model proposing a revision may see it."""
187
+
188
+ task_hash: Digest
189
+ task_label: NonEmptyString
190
+ public_prompt: str | None
191
+ subject_reply: str | None
192
+ reward: float
193
+ outcome: ImprovementOutcome
194
+ public_metrics: dict[str, float | None]
195
+ error_summary: str | None
196
+
197
+
198
+ class SkillImprovementContext(ProtocolModel):
199
+ """Everything a host agent is given to propose one Skill revision.
200
+
201
+ The four fingerprints decisions document 0007 R2 requires are all here:
202
+ the source Skill's root digest, its entrypoint file digest, the source run
203
+ ID, and the source report digest. They are what lets a consumer resolve
204
+ the run-owned verified snapshot, re-verify it, and bind a proposal to the
205
+ exact Skill and the exact result it was made against. None of them is the
206
+ Skill's text: the text is read through ``techtree uplift skill-source``,
207
+ which verifies it against these same digests, so a context that travelled
208
+ somewhere cannot carry a Skill nobody checked.
209
+ """
210
+
211
+ schema_version: Literal["techtree.skill-improvement-context.v1"]
212
+ source_run_id: NonEmptyString
213
+ source_report_digest: Digest
214
+ campaign_spec_digest: Digest
215
+ parent_skill_digest: Digest
216
+ parent_skill_entrypoint_digest: Digest
217
+ data_policy_digest: Digest
218
+ objective: NonEmptyString
219
+ current_result: PrimaryUpliftResult
220
+ examples: list[ImprovementExample]
221
+ constraints: list[NonEmptyString]
222
+ prohibited_material: list[NonEmptyString]
223
+
224
+
225
+ def build_improvement_context(
226
+ *,
227
+ report: UpliftReport,
228
+ candidate_receipts: Sequence[EpisodeReceipt],
229
+ baseline_receipts: Sequence[EpisodeReceipt],
230
+ campaign: CampaignSpec,
231
+ parent_skill: SkillArtifact,
232
+ task_public_projection: TaskPublicProjectionProvider = hash_only_projection,
233
+ ) -> SkillImprovementContext:
234
+ """Build the sanitized local context for one completed run.
235
+
236
+ Every number comes from the signed report or the signed receipts, so two
237
+ builds over one run produce identical bytes. Ordering is spec section
238
+ 7.18's: regressions, then the tasks the candidate still fails, then the
239
+ narrowest wins, and a small contrast sample of what already works.
240
+ """
241
+ _require(
242
+ report.campaign_spec_digest == digest_object(campaign),
243
+ "this report is not a report of the Campaign the context would "
244
+ "describe, so the two are about different experiments",
245
+ expected=report.campaign_spec_digest,
246
+ computed=digest_object(campaign),
247
+ )
248
+ _require(
249
+ bool(report.task_deltas),
250
+ "this run compared no tasks, so there is nothing to improve against",
251
+ run_id=report.run_id,
252
+ )
253
+
254
+ by_task = {
255
+ receipt.task_hash: receipt
256
+ for receipt in candidate_receipts
257
+ if receipt.run_id == report.run_id
258
+ }
259
+ examples = _select(
260
+ [
261
+ _example(
262
+ delta=delta,
263
+ position=position,
264
+ receipt=by_task.get(delta.task_hash),
265
+ projection=task_public_projection(
266
+ task_hash=delta.task_hash, position=position
267
+ ),
268
+ )
269
+ for position, delta in enumerate(report.task_deltas)
270
+ ]
271
+ )
272
+
273
+ context = SkillImprovementContext(
274
+ schema_version=IMPROVEMENT_CONTEXT_SCHEMA_VERSION,
275
+ source_run_id=report.run_id,
276
+ source_report_digest=digest_object(report),
277
+ campaign_spec_digest=report.campaign_spec_digest,
278
+ parent_skill_digest=parent_skill.root_digest,
279
+ parent_skill_entrypoint_digest=entrypoint_digest(parent_skill),
280
+ data_policy_digest=report.data_policy_digest,
281
+ objective=_objective(campaign, report.primary_result),
282
+ current_result=report.primary_result,
283
+ examples=examples,
284
+ constraints=list(REVISION_CONSTRAINTS),
285
+ prohibited_material=list(PROHIBITED_MATERIAL),
286
+ )
287
+ ensure_no_hidden_material(context)
288
+ return context
289
+
290
+
291
+ def entrypoint_digest(skill: SkillArtifact) -> Digest:
292
+ """Return the digest of the one file a Skill is read through.
293
+
294
+ Decisions document 0007 R2 pins the entrypoint separately from the tree it
295
+ belongs to, because the text a host model is shown is that one file. The
296
+ root digest says the whole Skill is unchanged; this says the file whose
297
+ bytes were actually read is the file that was measured.
298
+ """
299
+ for entry in skill.files:
300
+ if entry.path == SKILL_ENTRY_FILE:
301
+ return entry.digest
302
+ # Unreachable through the model, which refuses a Skill without its entry
303
+ # file. Stated anyway, because the alternative is an index that silently
304
+ # picks the wrong file the day that validator changes.
305
+ raise ValidationError(
306
+ f"the Skill this context describes lists no {SKILL_ENTRY_FILE}, so "
307
+ "there is no entrypoint to pin",
308
+ code=IMPROVEMENT_CONTEXT_INVALID,
309
+ details={"skill": skill.root_digest},
310
+ )
311
+
312
+
313
+ def ensure_no_hidden_material(context: SkillImprovementContext) -> None:
314
+ """Check every free-text field before a context is handed to anything.
315
+
316
+ The shape already keeps hidden material out; this keeps a credential or an
317
+ absolute path out of the free text the shape still allows, exactly as the
318
+ presentation payload is checked before it is rendered.
319
+ """
320
+ for label, value in _free_text(context):
321
+ _forbid(label, value)
322
+
323
+
324
+ # ---------------------------------------------------------------------------
325
+ # The pieces
326
+ # ---------------------------------------------------------------------------
327
+
328
+
329
+ def _example(
330
+ *,
331
+ delta: TaskDelta,
332
+ position: int,
333
+ receipt: EpisodeReceipt | None,
334
+ projection: TaskPublicProjection,
335
+ ) -> ImprovementExample:
336
+ """Describe one task from the signed record of what it scored.
337
+
338
+ A caller's projection is checked for credentials *before* it is shortened,
339
+ because shortening a leaked key would hide the fact that one reached this
340
+ far. Bounding and flattening happen after, and only to text that has
341
+ already been found clean.
342
+ """
343
+ _forbid("task_label", projection.task_label)
344
+ if projection.public_prompt is not None:
345
+ _forbid("public_prompt", projection.public_prompt)
346
+
347
+ return ImprovementExample(
348
+ task_hash=delta.task_hash,
349
+ task_label=sanitize_label(projection.task_label),
350
+ public_prompt=(
351
+ None
352
+ if projection.public_prompt is None
353
+ else sanitize_label(projection.public_prompt, maximum=600)
354
+ ),
355
+ # Always absent in this build. See this module's docstring: for a
356
+ # taskset scored by matching an answer, a correct reply is the expected
357
+ # answer, and spec section 7.18's exclusions control.
358
+ subject_reply=None,
359
+ reward=delta.candidate_reward,
360
+ outcome=_outcome(delta),
361
+ public_metrics={
362
+ "baseline_reward": delta.baseline_reward,
363
+ "delta": delta.delta,
364
+ **_recorded_metrics(receipt),
365
+ },
366
+ error_summary=_error_summary(receipt),
367
+ )
368
+
369
+
370
+ def _outcome(delta: TaskDelta) -> ImprovementOutcome:
371
+ """Return which way one task moved, and whether it stands where it should.
372
+
373
+ A reward of zero is the one value every Verifiers reward function agrees
374
+ means "earned nothing here", so it is what separates a task that stably
375
+ works from one that stably does not. Nothing else about the reward's scale
376
+ is assumed, because nothing else about it is declared.
377
+ """
378
+ if delta.delta < 0:
379
+ return "regressed"
380
+ if delta.delta > 0:
381
+ return "improved"
382
+ return "stable_success" if delta.candidate_reward > 0.0 else "stable_failure"
383
+
384
+
385
+ def _recorded_metrics(receipt: EpisodeReceipt | None) -> dict[str, float | None]:
386
+ """Return the metrics this task's subject traces recorded, if any.
387
+
388
+ These are the evaluation's own named measurements, which the report already
389
+ carries the rewards half of. A metric whose name would collide with the two
390
+ this context computes is left out rather than silently overwriting one.
391
+ """
392
+ if receipt is None:
393
+ return {}
394
+ recorded: dict[str, float | None] = {}
395
+ for traces in receipt.named_traces.values():
396
+ for trace in traces:
397
+ for name, value in sorted(trace.metrics.items()):
398
+ if name in ("baseline_reward", "delta"):
399
+ continue
400
+ recorded[sanitize_label(name, maximum=64)] = value
401
+ return recorded
402
+
403
+
404
+ def _error_summary(receipt: EpisodeReceipt | None) -> str | None:
405
+ """Say why a task's score is not usable evidence, when it is not.
406
+
407
+ The engine's recorded exception text lives in the normalized evaluation
408
+ output, which the run's proof does not cover and this context does not
409
+ read. What the signed receipt does say is the status its score carries,
410
+ and that is the part a reviser can act on.
411
+ """
412
+ if receipt is None:
413
+ return "no receipt was recorded for this task in the candidate variant"
414
+ if receipt.score_status is ScoreStatus.VALID:
415
+ return None
416
+ return f"the recorded score for this task is {receipt.score_status.value}"
417
+
418
+
419
+ _OUTCOME_RANK: Final[dict[ImprovementOutcome, int]] = {
420
+ "regressed": 0,
421
+ "stable_failure": 1,
422
+ "improved": 2,
423
+ "stable_success": 3,
424
+ }
425
+
426
+
427
+ def _select(examples: Sequence[ImprovementExample]) -> list[ImprovementExample]:
428
+ """Order and bound the tasks a reviser is shown. Spec section 7.18.
429
+
430
+ Regressions come first and worst-first, because a regression is the one
431
+ outcome that says the Skill actively hurt. Then the tasks the candidate
432
+ still fails. Then the wins by the narrowest margin, which are the ones a
433
+ small change could lose. Stable successes ride along only as contrast and
434
+ only a few of them.
435
+
436
+ The list is *emitted* in that order too, not restored to Campaign order. A
437
+ model reads from the top and a bounded list read from the top should start
438
+ with what went wrong. Committed order is recoverable from the report, which
439
+ carries every task; this list carries the ones worth looking at.
440
+ """
441
+ ranked = sorted(
442
+ enumerate(examples),
443
+ key=lambda item: (
444
+ _OUTCOME_RANK[item[1].outcome],
445
+ _within_group(item[1]),
446
+ item[0],
447
+ ),
448
+ )
449
+ chosen: list[ImprovementExample] = []
450
+ contrast = 0
451
+ for _, example in ranked:
452
+ if example.outcome == "stable_success":
453
+ if contrast >= EXAMPLE_CONTRAST_LIMIT:
454
+ continue
455
+ contrast += 1
456
+ chosen.append(example)
457
+ if len(chosen) >= EXAMPLE_LIMIT:
458
+ break
459
+ return chosen
460
+
461
+
462
+ def _within_group(example: ImprovementExample) -> float:
463
+ """Return the margin that orders one task inside its outcome group.
464
+
465
+ Regressions are worst first, so the most negative delta sorts first. Wins
466
+ are narrowest first, so the smallest positive delta sorts first. Both are
467
+ the same expression, which is why they are one line.
468
+ """
469
+ delta = example.public_metrics.get("delta")
470
+ return 0.0 if delta is None else delta
471
+
472
+
473
+ def _objective(campaign: CampaignSpec, result: PrimaryUpliftResult) -> str:
474
+ """State, in one sentence a model can act on, what a revision has to beat.
475
+
476
+ The margin clause is dropped when the Campaign declares no margin, because
477
+ "by at least 0.000" is a requirement that is not one, and a model told to
478
+ satisfy it would be being told something false about the contract.
479
+ """
480
+ scoring = campaign.scoring
481
+ threshold = scoring.minimum_absolute_delta
482
+ tasks = len(campaign.taskset.membership.ordered_task_hashes)
483
+ margin = "" if threshold <= 0.0 else f", by at least {threshold:.3f}"
484
+ return sanitize_label(
485
+ f"Revise the Skill so that mean {scoring.primary_reward} over the same "
486
+ f"{tasks} committed tasks rises above {result.candidate_mean:.3f}"
487
+ f"{margin}, without changing anything else about the experiment.",
488
+ maximum=400,
489
+ )
490
+
491
+
492
+ def _free_text(context: SkillImprovementContext) -> list[tuple[str, str]]:
493
+ """Return every free-text field a context carries, with its name."""
494
+ found = [
495
+ ("objective", context.objective),
496
+ *(
497
+ (f"constraints[{index}]", value)
498
+ for index, value in enumerate(context.constraints)
499
+ ),
500
+ *(
501
+ (f"prohibited_material[{index}]", value)
502
+ for index, value in enumerate(context.prohibited_material)
503
+ ),
504
+ ]
505
+ for index, example in enumerate(context.examples):
506
+ found.append((f"examples[{index}].task_label", example.task_label))
507
+ for name in ("public_prompt", "subject_reply", "error_summary"):
508
+ value = getattr(example, name)
509
+ if value is not None:
510
+ found.append((f"examples[{index}].{name}", value))
511
+ found.extend(
512
+ (f"examples[{index}].public_metrics[{key}]", key)
513
+ for key in example.public_metrics
514
+ )
515
+ return found
516
+
517
+
518
+ def _forbid(label: str, value: str) -> None:
519
+ """Refuse one string that carries material a context may not carry.
520
+
521
+ Refusing rather than quietly editing is the same rule spec section 7.17
522
+ applies to a rendering, for the same reason: a context that swallowed
523
+ somebody's home directory would hide the fact that one reached this far.
524
+ """
525
+ try:
526
+ ensure_no_control_or_local_path(value, field=label)
527
+ except ValidationError as error:
528
+ raise ValidationError(
529
+ "a value bound for an improvement context carries material the "
530
+ f"context excludes: {error.message}",
531
+ code=IMPROVEMENT_CONTEXT_FORBIDDEN_MATERIAL,
532
+ details={"field": label},
533
+ ) from error
534
+
535
+
536
+ def _require(condition: bool, message: str, **details: str) -> None:
537
+ """Raise a typed refusal unless the condition holds."""
538
+ if condition:
539
+ return
540
+ raise ValidationError(
541
+ message,
542
+ code=IMPROVEMENT_CONTEXT_INVALID,
543
+ details=dict(details),
544
+ )