techtree 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. techtree/__init__.py +35 -0
  2. techtree/__main__.py +14 -0
  3. techtree/canonical.py +239 -0
  4. techtree/catalog/__init__.py +25 -0
  5. techtree/catalog/repository.py +400 -0
  6. techtree/catalog/service.py +419 -0
  7. techtree/cli/__init__.py +1 -0
  8. techtree/cli/app.py +416 -0
  9. techtree/cli/commands/__init__.py +1 -0
  10. techtree/cli/commands/climb.py +1223 -0
  11. techtree/cli/commands/doctor.py +147 -0
  12. techtree/cli/commands/engine.py +207 -0
  13. techtree/cli/commands/proof.py +556 -0
  14. techtree/cli/commands/publish.py +447 -0
  15. techtree/cli/commands/release.py +303 -0
  16. techtree/cli/commands/run.py +1067 -0
  17. techtree/cli/commands/setup.py +181 -0
  18. techtree/cli/commands/skill.py +221 -0
  19. techtree/cli/commands/uplift.py +698 -0
  20. techtree/cli/commands/withdraw.py +212 -0
  21. techtree/cli/confirm.py +47 -0
  22. techtree/cli/context.py +96 -0
  23. techtree/cli/invoke.py +220 -0
  24. techtree/cli/output.py +280 -0
  25. techtree/constants.py +138 -0
  26. techtree/crypto.py +128 -0
  27. techtree/doctor/__init__.py +1 -0
  28. techtree/doctor/checks.py +675 -0
  29. techtree/doctor/execution_checks.py +435 -0
  30. techtree/doctor/service.py +326 -0
  31. techtree/drafts/__init__.py +32 -0
  32. techtree/drafts/source.py +146 -0
  33. techtree/drafts/store.py +992 -0
  34. techtree/engines/__init__.py +1 -0
  35. techtree/engines/bundle.py +251 -0
  36. techtree/engines/installer.py +679 -0
  37. techtree/engines/registry.py +235 -0
  38. techtree/engines/runner.py +170 -0
  39. techtree/errors.py +262 -0
  40. techtree/fs.py +234 -0
  41. techtree/harness.py +108 -0
  42. techtree/identity/__init__.py +41 -0
  43. techtree/identity/models.py +113 -0
  44. techtree/identity/service.py +199 -0
  45. techtree/identity/store.py +263 -0
  46. techtree/ids.py +85 -0
  47. techtree/manifests/__init__.py +39 -0
  48. techtree/manifests/builder.py +433 -0
  49. techtree/manifests/compare.py +376 -0
  50. techtree/models/__init__.py +282 -0
  51. techtree/models/base.py +201 -0
  52. techtree/models/campaign.py +484 -0
  53. techtree/models/catalog.py +227 -0
  54. techtree/models/cli.py +151 -0
  55. techtree/models/climb.py +254 -0
  56. techtree/models/data_policy.py +130 -0
  57. techtree/models/engine.py +156 -0
  58. techtree/models/episode_receipt.py +130 -0
  59. techtree/models/evaluation_backend.py +113 -0
  60. techtree/models/experiment.py +154 -0
  61. techtree/models/run.py +214 -0
  62. techtree/models/skill.py +156 -0
  63. techtree/models/uplift_report.py +158 -0
  64. techtree/models/validation.py +299 -0
  65. techtree/paths.py +116 -0
  66. techtree/presentation/__init__.py +31 -0
  67. techtree/presentation/build.py +1242 -0
  68. techtree/presentation/compact.py +246 -0
  69. techtree/presentation/evidence.py +169 -0
  70. techtree/presentation/models.py +358 -0
  71. techtree/presentation/rich.py +312 -0
  72. techtree/presentation/sanitize.py +156 -0
  73. techtree/publication/__init__.py +44 -0
  74. techtree/publication/address.py +180 -0
  75. techtree/publication/coordinates.py +26 -0
  76. techtree/publication/journal.py +212 -0
  77. techtree/publication/keccak.py +183 -0
  78. techtree/publication/models.py +209 -0
  79. techtree/publication/offer.py +35 -0
  80. techtree/publication/service.py +618 -0
  81. techtree/publication/transport.py +296 -0
  82. techtree/publication/verify.py +242 -0
  83. techtree/publication/withdraw.py +156 -0
  84. techtree/py.typed +0 -0
  85. techtree/receipts/__init__.py +52 -0
  86. techtree/receipts/bundle.py +578 -0
  87. techtree/receipts/compare.py +1065 -0
  88. techtree/receipts/episode.py +672 -0
  89. techtree/receipts/execution.py +630 -0
  90. techtree/receipts/observed.py +474 -0
  91. techtree/receipts/set.py +336 -0
  92. techtree/receipts/uplift.py +655 -0
  93. techtree/receipts/verify.py +1055 -0
  94. techtree/release/__init__.py +9 -0
  95. techtree/release/bootstrap.py +509 -0
  96. techtree/release/checks.py +376 -0
  97. techtree/release/document.py +125 -0
  98. techtree/release/generate.py +221 -0
  99. techtree/release/models.py +293 -0
  100. techtree/release/provenance.py +109 -0
  101. techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
  102. techtree/resources/catalog/catalog.json +32 -0
  103. techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
  104. techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
  105. techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
  106. techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
  107. techtree/resources/engines/default/engine.json +20 -0
  108. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
  109. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
  110. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
  111. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
  112. techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
  113. techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
  114. techtree/resources/engines/default/pyproject.toml +23 -0
  115. techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
  116. techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
  117. techtree/resources/engines/default/tools/normalize_validation.py +222 -0
  118. techtree/resources/engines/default/uv.lock +1758 -0
  119. techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
  120. techtree/resources/release/build-provenance.json +4 -0
  121. techtree/resources/release/release-core.json +24 -0
  122. techtree/runs/__init__.py +31 -0
  123. techtree/runs/artifacts.py +750 -0
  124. techtree/runs/child_registry.py +228 -0
  125. techtree/runs/events.py +478 -0
  126. techtree/runs/executor.py +140 -0
  127. techtree/runs/fake.py +741 -0
  128. techtree/runs/launcher.py +253 -0
  129. techtree/runs/machine.py +489 -0
  130. techtree/runs/real.py +789 -0
  131. techtree/runs/service.py +616 -0
  132. techtree/runs/store.py +555 -0
  133. techtree/runs/validation.py +259 -0
  134. techtree/runs/variants.py +684 -0
  135. techtree/settings.py +143 -0
  136. techtree/skills/__init__.py +14 -0
  137. techtree/skills/archive.py +282 -0
  138. techtree/skills/policy.py +62 -0
  139. techtree/skills/scanner.py +394 -0
  140. techtree/skills/service.py +752 -0
  141. techtree/skills/starter.py +434 -0
  142. techtree/tasksets/__init__.py +1 -0
  143. techtree/tasksets/membership.py +269 -0
  144. techtree/tasksets/provider.py +207 -0
  145. techtree/tasksets/resolver.py +311 -0
  146. techtree/tasksets/service.py +484 -0
  147. techtree/tasksets/verifiers_cli.py +538 -0
  148. techtree/uplift/__init__.py +20 -0
  149. techtree/uplift/context.py +544 -0
  150. techtree/uplift/derive.py +203 -0
  151. techtree/uplift/public_tasks.py +151 -0
  152. techtree/uplift/service.py +719 -0
  153. techtree/uplift/source.py +160 -0
  154. techtree/verifiers/__init__.py +31 -0
  155. techtree/verifiers/budget.py +219 -0
  156. techtree/verifiers/child.py +633 -0
  157. techtree/verifiers/compiler.py +432 -0
  158. techtree/verifiers/config.py +365 -0
  159. techtree/verifiers/credentials.py +321 -0
  160. techtree/verifiers/image.py +126 -0
  161. techtree/verifiers/models.py +527 -0
  162. techtree/verifiers/outputs.py +368 -0
  163. techtree/verifiers/progress.py +192 -0
  164. techtree/verifiers/supervisor.py +341 -0
  165. techtree/verifiers/verify.py +782 -0
  166. techtree/version.py +39 -0
  167. techtree/worker/__init__.py +18 -0
  168. techtree/worker/execute.py +487 -0
  169. techtree/worker/main.py +57 -0
  170. techtree-0.1.0.dist-info/METADATA +344 -0
  171. techtree-0.1.0.dist-info/RECORD +174 -0
  172. techtree-0.1.0.dist-info/WHEEL +4 -0
  173. techtree-0.1.0.dist-info/entry_points.txt +3 -0
  174. techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,1065 @@
1
+ """Proving the two variants were one experiment. Spec section 7.9.
2
+
3
+ :mod:`techtree.manifests.compare` proves that the two documents a run was
4
+ prepared from differ only where the Campaign permits. That is a claim about
5
+ *intentions*. This module makes the harder claim: that the two executions the
6
+ run actually performed differ only there too.
7
+
8
+ The difference matters because everything between a manifest and a container
9
+ can drift. A provider can route a model identifier somewhere else, a harness can
10
+ resolve a different version, a daemon can hand back a different image, a
11
+ configuration can pick up a sampling parameter nobody declared. None of that is
12
+ visible in a manifest, and all of it would be reported as uplift.
13
+
14
+ So the comparison runs over three sources at once:
15
+
16
+ *What was declared* — the two ``ExperimentManifest`` documents and the Campaign
17
+ they were derived from, checked field by field and then diffed whole.
18
+
19
+ *What was observed* — one fingerprint per variant, computed by
20
+ :mod:`techtree.receipts.observed` from the traces the runtime wrote and the
21
+ configuration the engine resolved, checked against the manifest that variant was
22
+ supposed to be and against the other variant's fingerprint.
23
+
24
+ *What was joined* — one receipt per committed task on each side, paired by task
25
+ hash in the order the TasksetLock fixes, so the aggregation downstream cannot
26
+ pair by arrival order or quietly drop a task.
27
+
28
+ Four decisions are worth stating.
29
+
30
+ *The whole tool surface cannot be required to match, and here is exactly what
31
+ may not.* Mounting a Skill changes what the subject is offered, because Hermes
32
+ 0.19.0 renders the index of visible Skills into the description of its own
33
+ Skill-management tool. Across two recorded probes of the same Campaign the
34
+ fifteen tool names, fifteen parameter schemas and fourteen of the fifteen
35
+ descriptions are byte-identical, and ``skill_manage``'s description differs. A
36
+ check that required one tool-inventory digest would therefore reject every
37
+ clean run. What is permitted is the exact derived difference decisions document
38
+ 0007 ratified: the tool names and parameter schemas of the pinned harness
39
+ conformance fixture (:mod:`techtree.harness`), unchanged on both sides, with at
40
+ most one differing description and only on :data:`SKILL_INDEX_TOOL`. A second
41
+ differing description, a differing schema, a differing description anywhere
42
+ else, or any departure from the fixture's own surface is a violation — the
43
+ fixture is what stops a moved harness pin from being read as the Skill index
44
+ doing what the Skill index does.
45
+
46
+ *A weaker claim is a warning, never silence and never a failure.* One thing
47
+ about a real run is honestly unverifiable here: a provider that publishes no
48
+ model revision cannot be pinned to one. It is recorded as a warning, which is
49
+ what
50
+ :class:`~techtree.models.uplift_report.ComparisonStatus.CONTROLLED_WITH_WARNINGS`
51
+ exists for, and it is never allowed to absorb an actual mismatch: a model that
52
+ differs is a failure whether or not its revision was discoverable. What ran in
53
+ the container used to be a warning of the same kind and is a check now — see
54
+ :func:`_runtime_pin_checks`.
55
+
56
+ *This function reports; it does not refuse.* Like
57
+ :func:`~techtree.manifests.compare.compare_manifests`, it returns what it found
58
+ so that an operator can read every check. Turning an invalid comparison into a
59
+ refusal is :mod:`techtree.receipts.uplift`'s job, at the point where a report
60
+ would otherwise be written.
61
+
62
+ *The observed inputs are richer than section 7.9's signature.* The specification
63
+ sketches ``compare_real_variants`` as taking receipts alone. A frozen
64
+ ``EpisodeReceipt`` carries rewards and lineage and no configuration at all, so
65
+ the observed side is passed explicitly as :class:`ObservedVariant`, built by
66
+ :func:`observe_variant` from the same evidence the receipts were built from.
67
+ The schedule is passed for the same reason: which schedule ran, and how far
68
+ apart the two children were, is a fact about the execution rather than about
69
+ either variant.
70
+ """
71
+
72
+ from __future__ import annotations
73
+
74
+ from collections.abc import Mapping, Sequence
75
+ from dataclasses import dataclass
76
+ from typing import Any, Final, Literal, Self
77
+
78
+ from pydantic import Field, model_validator
79
+
80
+ from techtree.canonical import digest_object, validate_digest
81
+ from techtree.harness import harness_conformance
82
+ from techtree.manifests.compare import compare_manifests
83
+ from techtree.models.base import Digest, JsonValue, NonEmptyString, ProtocolModel
84
+ from techtree.models.campaign import (
85
+ SUBJECT_AGENT,
86
+ AgentSpec,
87
+ CampaignSpec,
88
+ MutationKind,
89
+ RuntimeSpec,
90
+ VariantSchedule,
91
+ )
92
+ from techtree.models.episode_receipt import EpisodeReceipt
93
+ from techtree.models.experiment import (
94
+ ExperimentManifest,
95
+ ExperimentVariant,
96
+ ManifestComparison,
97
+ )
98
+ from techtree.models.uplift_report import ComparisonStatus
99
+ from techtree.models.validation import TasksetLock
100
+ from techtree.receipts.observed import (
101
+ ObservedSubjectConfiguration,
102
+ observed_from_episodes,
103
+ )
104
+ from techtree.tasksets.membership import membership_digest
105
+ from techtree.verifiers.models import (
106
+ ChildProcessOutcome,
107
+ NormalizedTool,
108
+ VariantExecutionResult,
109
+ )
110
+
111
+ __all__ = [
112
+ "COMPARISON_INVALID",
113
+ "MODEL_REVISION_UNDISCOVERABLE",
114
+ "SKILL_INDEX_TOOL",
115
+ "ComparisonCheck",
116
+ "ObservedVariant",
117
+ "PairedReceiptRow",
118
+ "RealComparisonResult",
119
+ "ScheduleObservation",
120
+ "compare_real_variants",
121
+ "observe_variant",
122
+ "weaker_claim_warnings",
123
+ ]
124
+
125
+ #: Stable error code. Spec section 15. Raised by whoever turns an invalid
126
+ #: comparison into a refusal, never by this module.
127
+ COMPARISON_INVALID: Final = "comparison_invalid"
128
+
129
+ #: The one tool whose *description* two controlled variants may disagree about.
130
+ #:
131
+ #: Hermes 0.19.0 lists the Skills currently visible to the subject inside this
132
+ #: tool's description, so inserting or replacing a Skill necessarily changes it.
133
+ #: It is a consequence of the mutation rather than a second difference, and it
134
+ #: was measured on the recorded probes rather than assumed: see
135
+ #: ``tests/fixtures/receipts/support.py`` and
136
+ #: ``test_inserting_a_skill_changes_exactly_one_tool_description``.
137
+ SKILL_INDEX_TOOL: Final = "skill_manage"
138
+
139
+ #: The one weaker-claim warning this build can record, named so that a reader
140
+ #: of a result can be told which coordinate it is about rather than only that
141
+ #: there was one. :func:`weaker_claim_warnings` is what decides whether a run
142
+ #: has it.
143
+ MODEL_REVISION_UNDISCOVERABLE: Final = "model_revision_discoverable"
144
+
145
+ _PASSED: Final = "passed"
146
+ _FAILED: Final = "failed"
147
+ _WARNING: Final = "warning"
148
+
149
+
150
+ class ComparisonCheck(ProtocolModel):
151
+ """One named question about a comparison, and its answer.
152
+
153
+ The same shape as
154
+ :class:`~techtree.models.validation.ValidationCheck` and
155
+ :class:`~techtree.verifiers.models.ExecutionCheck`, and for the same
156
+ reason: a reader wants an ordered list of named verdicts, not an exception
157
+ that stops at whichever rule happened to fail first.
158
+ """
159
+
160
+ id: NonEmptyString
161
+ status: Literal["passed", "failed", "warning"]
162
+ detail: NonEmptyString
163
+
164
+
165
+ class ScheduleObservation(ProtocolModel):
166
+ """How the two variants were placed in time. Spec section 7.9.
167
+
168
+ Operational metadata, deliberately kept apart from the scientific checks: a
169
+ wide launch skew does not invalidate a comparison, it just means the two
170
+ sides saw the provider at more distant moments, and a reader deciding how
171
+ much that matters needs the number rather than a verdict about it.
172
+ """
173
+
174
+ schedule: VariantSchedule
175
+ start_skew_seconds: float = Field(ge=0.0)
176
+ completion_window_seconds: float = Field(ge=0.0)
177
+ overlapped: bool
178
+
179
+
180
+ class PairedReceiptRow(ProtocolModel):
181
+ """One committed task, and the two receipts that scored it."""
182
+
183
+ position: int = Field(ge=0)
184
+ task_hash: Digest
185
+ baseline_receipt_id: NonEmptyString
186
+ baseline_receipt_digest: Digest
187
+ candidate_receipt_id: NonEmptyString
188
+ candidate_receipt_digest: Digest
189
+
190
+
191
+ class RealComparisonResult(ProtocolModel):
192
+ """Whether the two executions were one experiment, and how it was checked.
193
+
194
+ A local object, versioned by the run directory it lives in rather than by
195
+ the protocol: the frozen v0.1 schema has no controlled-comparison document,
196
+ and inventing one here would be a protocol amendment nobody ratified.
197
+ """
198
+
199
+ status: ComparisonStatus
200
+ mutation_kind: MutationKind
201
+ manifest_comparison: ManifestComparison
202
+ ordered_task_hashes: list[Digest]
203
+ rows: list[PairedReceiptRow]
204
+ checks: list[ComparisonCheck]
205
+ schedule: ScheduleObservation
206
+ baseline_observed: ObservedSubjectConfiguration
207
+ candidate_observed: ObservedSubjectConfiguration
208
+
209
+ @property
210
+ def failures(self) -> list[ComparisonCheck]:
211
+ """Return every check that found a scientific invariant broken."""
212
+ return [check for check in self.checks if check.status == _FAILED]
213
+
214
+ @property
215
+ def warnings(self) -> list[ComparisonCheck]:
216
+ """Return every check that found a weaker claim than it wanted."""
217
+ return [check for check in self.checks if check.status == _WARNING]
218
+
219
+ @property
220
+ def controlled(self) -> bool:
221
+ """Whether this comparison may carry a scientific result."""
222
+ return self.status in (
223
+ ComparisonStatus.CONTROLLED,
224
+ ComparisonStatus.CONTROLLED_WITH_WARNINGS,
225
+ )
226
+
227
+ @model_validator(mode="after")
228
+ def _check_the_status_is_the_one_the_checks_support(self) -> Self:
229
+ """Reject a result whose headline disagrees with its own checks."""
230
+ expected = _status_for(self.checks)
231
+ if self.status is not expected:
232
+ raise ValueError(
233
+ f"a comparison whose checks are {expected.value} cannot report "
234
+ f"{self.status.value}"
235
+ )
236
+ if self.status is not ComparisonStatus.INVALID and not self.rows:
237
+ raise ValueError("a controlled comparison pairs at least one task")
238
+ return self
239
+
240
+
241
+ @dataclass(frozen=True)
242
+ class ObservedVariant:
243
+ """Everything about one variant that was measured rather than declared.
244
+
245
+ :class:`~techtree.receipts.observed.ObservedSubjectConfiguration` is the
246
+ fingerprint, and it commits the tool inventory to a single digest. A
247
+ controlled comparison has to look inside that digest — one tool description
248
+ is allowed to differ — so the tools travel beside it, together with the
249
+ effective sampling table the fingerprint also summarizes, the tasks this
250
+ variant actually scored, and the operational envelope its child recorded.
251
+ """
252
+
253
+ variant: ExperimentVariant
254
+ configuration: ObservedSubjectConfiguration
255
+ tools: list[NormalizedTool]
256
+ sampling: dict[str, JsonValue]
257
+ ordered_task_hashes: list[Digest]
258
+ episode_count: int
259
+ child_outcome: ChildProcessOutcome
260
+
261
+
262
+ def observe_variant(
263
+ *,
264
+ result: VariantExecutionResult,
265
+ resolved_config: Mapping[str, Any],
266
+ runtime: RuntimeSpec,
267
+ ) -> ObservedVariant:
268
+ """Fingerprint one executed variant from its own evidence.
269
+
270
+ Raises whatever :func:`~techtree.receipts.observed.observed_from_episodes`
271
+ raises: a variant whose own rollouts disagree about what they ran has no
272
+ single observed configuration, and there is nothing to compare.
273
+ """
274
+ configuration = observed_from_episodes(
275
+ result.episodes,
276
+ resolved_config=resolved_config,
277
+ image_resolution=result.image_resolution,
278
+ runtime=runtime,
279
+ )
280
+ # The same reference rollout the fingerprint was taken from, chosen the
281
+ # same way: every rollout of one variant has already been required to agree
282
+ # with every other, so the first one describes all of them.
283
+ reference = next(trace for episode in result.episodes for trace in episode.traces)
284
+ return ObservedVariant(
285
+ variant=ExperimentVariant(result.variant.value),
286
+ configuration=configuration,
287
+ tools=sorted(reference.tools, key=lambda tool: tool.name),
288
+ sampling=dict(reference.sampling),
289
+ ordered_task_hashes=[
290
+ validate_digest(episode.task_hash) for episode in result.episodes
291
+ ],
292
+ episode_count=len(result.episodes),
293
+ child_outcome=result.child_outcome,
294
+ )
295
+
296
+
297
+ def compare_real_variants(
298
+ *,
299
+ campaign: CampaignSpec,
300
+ baseline_manifest: ExperimentManifest,
301
+ candidate_manifest: ExperimentManifest,
302
+ prepared_manifest_comparison: ManifestComparison,
303
+ baseline_receipts: Sequence[EpisodeReceipt],
304
+ candidate_receipts: Sequence[EpisodeReceipt],
305
+ taskset_lock: TasksetLock,
306
+ baseline_observed: ObservedVariant,
307
+ candidate_observed: ObservedVariant,
308
+ schedule: VariantSchedule,
309
+ ) -> RealComparisonResult:
310
+ """Verify declared and observed control and return the paired rows."""
311
+ committed = [validate_digest(value) for value in taskset_lock.ordered_task_hashes]
312
+ recomputed = compare_manifests(
313
+ baseline_manifest, candidate_manifest, campaign.mutation_contract
314
+ )
315
+
316
+ checks: list[ComparisonCheck] = [
317
+ *_declared_checks(
318
+ campaign=campaign,
319
+ baseline=baseline_manifest,
320
+ candidate=candidate_manifest,
321
+ prepared=prepared_manifest_comparison,
322
+ recomputed=recomputed,
323
+ taskset_lock=taskset_lock,
324
+ ),
325
+ *_observed_checks(
326
+ campaign=campaign,
327
+ baseline_manifest=baseline_manifest,
328
+ candidate_manifest=candidate_manifest,
329
+ baseline=baseline_observed,
330
+ candidate=candidate_observed,
331
+ committed=committed,
332
+ ),
333
+ ]
334
+ rows, pairing = _pair_receipts(
335
+ baseline_receipts=baseline_receipts,
336
+ candidate_receipts=candidate_receipts,
337
+ committed=committed,
338
+ )
339
+ checks.append(pairing)
340
+
341
+ observation = _observe_schedule(
342
+ schedule=schedule,
343
+ baseline=baseline_observed.child_outcome,
344
+ candidate=candidate_observed.child_outcome,
345
+ )
346
+ checks.append(
347
+ _check(
348
+ "schedule_recorded",
349
+ _PASSED,
350
+ f"the variants ran under {schedule.value}, launched "
351
+ f"{observation.start_skew_seconds:.3f}s apart and completed within "
352
+ f"{observation.completion_window_seconds:.3f}s",
353
+ )
354
+ )
355
+
356
+ status = _status_for(checks)
357
+ return RealComparisonResult(
358
+ status=status,
359
+ mutation_kind=campaign.mutation_contract.kind,
360
+ manifest_comparison=recomputed,
361
+ ordered_task_hashes=committed,
362
+ rows=[] if status is ComparisonStatus.INVALID else rows,
363
+ checks=checks,
364
+ schedule=observation,
365
+ baseline_observed=baseline_observed.configuration,
366
+ candidate_observed=candidate_observed.configuration,
367
+ )
368
+
369
+
370
+ # ---------------------------------------------------------------------------
371
+ # What the two documents declared
372
+ # ---------------------------------------------------------------------------
373
+
374
+
375
+ def _declared_checks(
376
+ *,
377
+ campaign: CampaignSpec,
378
+ baseline: ExperimentManifest,
379
+ candidate: ExperimentManifest,
380
+ prepared: ManifestComparison,
381
+ recomputed: ManifestComparison,
382
+ taskset_lock: TasksetLock,
383
+ ) -> list[ComparisonCheck]:
384
+ """Check every field spec section 7.9 requires the two variants to share.
385
+
386
+ The whole-configuration diff below already covers most of these, and they
387
+ are stated separately anyway. A diff can say that two documents disagree at
388
+ a pointer; it cannot say which disagreement was structurally impossible,
389
+ and an operator reading a failed comparison wants to be told "the two sides
390
+ name a different model", not "``/agents/subject/model/model_id``".
391
+ """
392
+ left = baseline.configuration
393
+ right = candidate.configuration
394
+ subject_left = _subject(baseline)
395
+ subject_right = _subject(candidate)
396
+
397
+ checks = [
398
+ _same(
399
+ "declared_campaign",
400
+ "Campaign",
401
+ baseline.campaign_spec_digest,
402
+ candidate.campaign_spec_digest,
403
+ ),
404
+ _same(
405
+ "declared_data_policy",
406
+ "DataPolicy",
407
+ left.data_policy_digest,
408
+ right.data_policy_digest,
409
+ ),
410
+ _same(
411
+ "declared_program_and_context",
412
+ "improvement program or public context",
413
+ (baseline.program_ref, baseline.public_context),
414
+ (candidate.program_ref, candidate.public_context),
415
+ ),
416
+ _same(
417
+ "declared_outcome_contract",
418
+ "OutcomeContract",
419
+ left.outcome_contract_digest,
420
+ right.outcome_contract_digest,
421
+ ),
422
+ _same(
423
+ "declared_evaluation_backend",
424
+ "evaluation backend",
425
+ left.evaluation_backend,
426
+ right.evaluation_backend,
427
+ ),
428
+ _same(
429
+ "declared_environment", "environment", left.environment, right.environment
430
+ ),
431
+ _same(
432
+ "declared_model", "subject model", subject_left.model, subject_right.model
433
+ ),
434
+ _same(
435
+ "declared_sampling",
436
+ "sampling",
437
+ subject_left.sampling,
438
+ subject_right.sampling,
439
+ ),
440
+ _same(
441
+ "declared_harness",
442
+ "harness identity, version or bundled-Skill setting",
443
+ _harness_identity(subject_left),
444
+ _harness_identity(subject_right),
445
+ ),
446
+ _same(
447
+ "declared_runtime",
448
+ "subject runtime or image",
449
+ subject_left.runtime,
450
+ subject_right.runtime,
451
+ ),
452
+ _same(
453
+ "declared_execution_contract",
454
+ "execution, scoring, evidence or budget contract",
455
+ (left.execution, left.scoring, left.evidence, left.budgets),
456
+ (right.execution, right.scoring, right.evidence, right.budgets),
457
+ ),
458
+ *_taskset_checks(campaign, baseline, candidate, taskset_lock),
459
+ *_manifest_comparison_checks(prepared, recomputed),
460
+ _mutation_check(campaign, subject_left, subject_right),
461
+ ]
462
+ return checks
463
+
464
+
465
+ def _taskset_checks(
466
+ campaign: CampaignSpec,
467
+ baseline: ExperimentManifest,
468
+ candidate: ExperimentManifest,
469
+ lock: TasksetLock,
470
+ ) -> list[ComparisonCheck]:
471
+ """Require one taskset, one committed membership, and one lock over it."""
472
+ committed = list(campaign.taskset.membership.ordered_task_hashes)
473
+ agreeing = (
474
+ baseline.configuration.taskset == campaign.taskset
475
+ and candidate.configuration.taskset == campaign.taskset
476
+ )
477
+ locked = (
478
+ list(lock.ordered_task_hashes) == committed
479
+ and lock.membership_digest == campaign.taskset.membership.membership_digest
480
+ and lock.membership_digest == membership_digest(committed)
481
+ and lock.taskset_ref == campaign.taskset.ref
482
+ and lock.task_count == len(committed)
483
+ )
484
+ return [
485
+ _check(
486
+ "declared_taskset",
487
+ _PASSED if agreeing else _FAILED,
488
+ (
489
+ "both variants commit to the Campaign's taskset and membership"
490
+ if agreeing
491
+ else "a variant commits to a different taskset or membership "
492
+ "than the Campaign"
493
+ ),
494
+ ),
495
+ _check(
496
+ "declared_taskset_lock",
497
+ _PASSED if locked else _FAILED,
498
+ (
499
+ f"the lock pins the {len(committed)} tasks the Campaign commits to"
500
+ if locked
501
+ else "the taskset lock does not pin the tasks, the membership "
502
+ "digest or the taskset the Campaign commits to"
503
+ ),
504
+ ),
505
+ ]
506
+
507
+
508
+ def _manifest_comparison_checks(
509
+ prepared: ManifestComparison, recomputed: ManifestComparison
510
+ ) -> list[ComparisonCheck]:
511
+ """Require the recomputed diff to be controlled and to be the prepared one."""
512
+ return [
513
+ _check(
514
+ "declared_only_skill_differs",
515
+ _PASSED if recomputed.controlled else _FAILED,
516
+ (
517
+ "the candidate configuration differs from the baseline only "
518
+ "where the mutation contract permits"
519
+ if recomputed.controlled
520
+ else "; ".join(recomputed.violations)
521
+ ),
522
+ ),
523
+ _check(
524
+ "declared_comparison_unchanged",
525
+ _PASSED if recomputed == prepared else _FAILED,
526
+ (
527
+ "the comparison recomputed from the executed manifests is the "
528
+ "one the run was prepared with"
529
+ if recomputed == prepared
530
+ else "the executed manifests do not produce the comparison this "
531
+ "run was prepared with"
532
+ ),
533
+ ),
534
+ ]
535
+
536
+
537
+ def _mutation_check(
538
+ campaign: CampaignSpec, baseline: AgentSpec, candidate: AgentSpec
539
+ ) -> ComparisonCheck:
540
+ """Hold the declared skill lists to the shape the mutation kind requires."""
541
+ kind = campaign.mutation_contract.kind
542
+ left = [reference.digest for reference in baseline.harness.skills]
543
+ right = [reference.digest for reference in candidate.harness.skills]
544
+
545
+ if kind is MutationKind.SKILL_INSERTION:
546
+ ok = not left and len(right) == 1
547
+ detail = (
548
+ "the baseline declares no Skill and the candidate declares one"
549
+ if ok
550
+ else f"a skill_insertion declares 0 then 1 Skill; got {len(left)} "
551
+ f"then {len(right)}"
552
+ )
553
+ else:
554
+ ok = len(left) == 1 and len(right) == 1 and left != right
555
+ detail = (
556
+ "the baseline and the candidate each declare one Skill, and they "
557
+ "are different Skills"
558
+ if ok
559
+ else "a skill_replacement declares one differing Skill on each side"
560
+ )
561
+ return _check(f"declared_mutation_{kind.value}", _PASSED if ok else _FAILED, detail)
562
+
563
+
564
+ # ---------------------------------------------------------------------------
565
+ # What the two executions did
566
+ # ---------------------------------------------------------------------------
567
+
568
+
569
+ def _observed_checks(
570
+ *,
571
+ campaign: CampaignSpec,
572
+ baseline_manifest: ExperimentManifest,
573
+ candidate_manifest: ExperimentManifest,
574
+ baseline: ObservedVariant,
575
+ candidate: ObservedVariant,
576
+ committed: Sequence[Digest],
577
+ ) -> list[ComparisonCheck]:
578
+ """Check what actually ran, against the other side and against the manifest."""
579
+ left = baseline.configuration
580
+ right = candidate.configuration
581
+
582
+ return [
583
+ _check(
584
+ "observed_task_order",
585
+ (
586
+ _PASSED
587
+ if baseline.ordered_task_hashes
588
+ == candidate.ordered_task_hashes
589
+ == list(committed)
590
+ else _FAILED
591
+ ),
592
+ (
593
+ "both variants scored the committed tasks in committed order"
594
+ if baseline.ordered_task_hashes
595
+ == candidate.ordered_task_hashes
596
+ == list(committed)
597
+ else "a variant scored a different set of tasks, or scored them "
598
+ "in an order the Campaign did not commit to"
599
+ ),
600
+ ),
601
+ _check(
602
+ "observed_episode_count",
603
+ (
604
+ _PASSED
605
+ if baseline.episode_count == candidate.episode_count == len(committed)
606
+ else _FAILED
607
+ ),
608
+ (
609
+ f"both variants recorded {len(committed)} episodes"
610
+ if baseline.episode_count == candidate.episode_count == len(committed)
611
+ else f"the variants recorded {baseline.episode_count} and "
612
+ f"{candidate.episode_count} episodes for {len(committed)} tasks"
613
+ ),
614
+ ),
615
+ _same("observed_model", "model", left.model_id, right.model_id),
616
+ _same(
617
+ "observed_sampling",
618
+ "effective sampling",
619
+ left.sampling_digest,
620
+ right.sampling_digest,
621
+ ),
622
+ _same(
623
+ "observed_harness",
624
+ "harness",
625
+ (left.harness_id, left.harness_version),
626
+ (right.harness_id, right.harness_version),
627
+ ),
628
+ _same(
629
+ "observed_bundled_skill",
630
+ "bundled-Skill setting",
631
+ left.use_bundled_skill,
632
+ right.use_bundled_skill,
633
+ ),
634
+ _same(
635
+ "observed_runtime_image",
636
+ "subject runtime or image",
637
+ (left.runtime_kind, left.runtime_image, left.runtime_image_index_digest),
638
+ (right.runtime_kind, right.runtime_image, right.runtime_image_index_digest),
639
+ ),
640
+ _same(
641
+ "observed_runtime_platform_digest",
642
+ "resolved platform-specific image digest",
643
+ (left.runtime_platform, left.runtime_image_platform_digest),
644
+ (right.runtime_platform, right.runtime_image_platform_digest),
645
+ ),
646
+ *_runtime_pin_checks(campaign, baseline, candidate),
647
+ _tool_surface_check(baseline, candidate),
648
+ _same(
649
+ "observed_reward_contract",
650
+ "reward names or weights",
651
+ left.reward_contract_digest,
652
+ right.reward_contract_digest,
653
+ ),
654
+ _same(
655
+ "observed_verifiers_build",
656
+ "Verifiers build",
657
+ (left.verifiers_version, left.verifiers_revision),
658
+ (right.verifiers_version, right.verifiers_revision),
659
+ ),
660
+ _declared_to_observed(baseline_manifest, baseline),
661
+ _declared_to_observed(candidate_manifest, candidate),
662
+ *weaker_claim_warnings(campaign),
663
+ ]
664
+
665
+
666
+ def _tool_surface_check(
667
+ baseline: ObservedVariant, candidate: ObservedVariant
668
+ ) -> ComparisonCheck:
669
+ """Permit the Skill index delta and nothing else. See :data:`SKILL_INDEX_TOOL`."""
670
+ left = {tool.name: tool for tool in baseline.tools}
671
+ right = {tool.name: tool for tool in candidate.tools}
672
+
673
+ departure = _conformance_departure(baseline, candidate)
674
+ if departure is not None:
675
+ return _check("observed_tool_inventory", _FAILED, departure)
676
+
677
+ if set(left) != set(right):
678
+ added = sorted(set(right) - set(left))
679
+ removed = sorted(set(left) - set(right))
680
+ return _check(
681
+ "observed_tool_inventory",
682
+ _FAILED,
683
+ "the two variants were offered different tools "
684
+ f"(added {added or 'nothing'}, removed {removed or 'nothing'})",
685
+ )
686
+
687
+ schemas = sorted(
688
+ name
689
+ for name in left
690
+ if left[name].parameters_digest != right[name].parameters_digest
691
+ )
692
+ if schemas:
693
+ return _check(
694
+ "observed_tool_inventory",
695
+ _FAILED,
696
+ f"the parameter schema of {', '.join(schemas)} differs between the "
697
+ "two variants",
698
+ )
699
+
700
+ descriptions = sorted(
701
+ name
702
+ for name in left
703
+ if left[name].description_digest != right[name].description_digest
704
+ )
705
+ if not descriptions:
706
+ return _check(
707
+ "observed_tool_inventory",
708
+ _PASSED,
709
+ f"both variants were offered the same {len(left)} tools, described "
710
+ "identically",
711
+ )
712
+ if descriptions == [SKILL_INDEX_TOOL]:
713
+ return _check(
714
+ "observed_tool_inventory",
715
+ _PASSED,
716
+ f"both variants were offered the same {len(left)} tools with the "
717
+ f"same schemas; only {SKILL_INDEX_TOOL}'s description differs, "
718
+ "which is where the harness lists the Skills the subject can see",
719
+ )
720
+ return _check(
721
+ "observed_tool_inventory",
722
+ _FAILED,
723
+ f"the description of {', '.join(descriptions)} differs between the two "
724
+ f"variants; only {SKILL_INDEX_TOOL}'s may",
725
+ )
726
+
727
+
728
+ def _conformance_departure(
729
+ baseline: ObservedVariant, candidate: ObservedVariant
730
+ ) -> str | None:
731
+ """Return why the offered tools are not the pinned harness's, or nothing.
732
+
733
+ Decisions document 0007 R9 item 4. A comparison sees one description
734
+ differing on one tool and cannot tell, from inside itself, whether that is
735
+ the Skill index doing its job or a harness that changed underneath the
736
+ Campaign. The pinned fixture is the outside evidence: it fixes the tool
737
+ count, the names and the parameter schemas of the harness build the
738
+ Campaign declares, so anything else is a departure and not a derived
739
+ difference.
740
+ """
741
+ for observed in (baseline, candidate):
742
+ configuration = observed.configuration
743
+ try:
744
+ pinned = harness_conformance(
745
+ configuration.harness_id, configuration.harness_version
746
+ )
747
+ except FileNotFoundError:
748
+ return (
749
+ f"no tool surface was ever recorded for "
750
+ f"{configuration.harness_id} {configuration.harness_version}, "
751
+ "so the difference between the two variants cannot be shown to "
752
+ "be the Skill index alone"
753
+ )
754
+ offered = sorted(tool.name for tool in observed.tools)
755
+ if offered != sorted(pinned.tool_names):
756
+ return (
757
+ f"the {observed.variant.value} was offered {len(offered)} tools "
758
+ f"where {configuration.harness_id} "
759
+ f"{configuration.harness_version} offers "
760
+ f"{len(pinned.tool_names)}, so the harness is not the one the "
761
+ "Campaign declares"
762
+ )
763
+ expected = pinned.parameters_by_tool
764
+ reshaped = sorted(
765
+ tool.name
766
+ for tool in observed.tools
767
+ if tool.parameters_digest != expected[tool.name]
768
+ )
769
+ if reshaped:
770
+ return (
771
+ f"the {observed.variant.value} was offered a different "
772
+ f"parameter schema for {', '.join(reshaped)} than "
773
+ f"{configuration.harness_id} {configuration.harness_version} "
774
+ "records"
775
+ )
776
+ if pinned.skill_index_tool != SKILL_INDEX_TOOL:
777
+ return (
778
+ f"{configuration.harness_id} {configuration.harness_version} "
779
+ f"renders its Skill index into {pinned.skill_index_tool}, not "
780
+ f"{SKILL_INDEX_TOOL}"
781
+ )
782
+ return None
783
+
784
+
785
+ def _declared_to_observed(
786
+ manifest: ExperimentManifest, observed: ObservedVariant
787
+ ) -> ComparisonCheck:
788
+ """Check one variant's execution against the manifest it was supposed to be.
789
+
790
+ Spec section 7.8 names this ``compare_declared_to_observed``. It lives here
791
+ because a check that only has meaning inside a controlled comparison should
792
+ be read beside the other checks of that comparison, and because
793
+ ``ComparisonCheck`` is this section's type.
794
+ """
795
+ subject = _subject(manifest)
796
+ configuration = observed.configuration
797
+ declared_sampling: dict[str, JsonValue] = {
798
+ "max_tokens": subject.sampling.max_tokens,
799
+ "temperature": subject.sampling.temperature,
800
+ }
801
+ mismatches = [
802
+ label
803
+ for label, declared, seen in (
804
+ ("model", subject.model.model_id, configuration.model_id),
805
+ ("harness", subject.harness.id, configuration.harness_id),
806
+ ("harness version", subject.harness.version, configuration.harness_version),
807
+ (
808
+ "bundled-Skill setting",
809
+ subject.harness.use_bundled_skill,
810
+ configuration.use_bundled_skill,
811
+ ),
812
+ ("runtime", subject.runtime.type, configuration.runtime_kind),
813
+ ("runtime image", subject.runtime.image, configuration.runtime_image),
814
+ (
815
+ "Skill list",
816
+ [reference.digest for reference in subject.harness.skills],
817
+ list(configuration.skill_root_digests),
818
+ ),
819
+ # Every parameter the engine resolved, and no parameter it did not:
820
+ # an undeclared sampling key is a difference the manifest never
821
+ # authorized, whichever variant picked it up.
822
+ ("sampling", declared_sampling, dict(observed.sampling)),
823
+ )
824
+ if declared != seen
825
+ ]
826
+ variant = observed.variant.value
827
+ return _check(
828
+ f"observed_matches_declared_{variant}",
829
+ _PASSED if not mismatches else _FAILED,
830
+ (
831
+ f"the {variant} executed the subject its manifest declares"
832
+ if not mismatches
833
+ else f"the {variant} executed a different {', '.join(mismatches)} "
834
+ "than its manifest declares"
835
+ ),
836
+ )
837
+
838
+
839
+ def _runtime_pin_checks(
840
+ campaign: CampaignSpec, baseline: ObservedVariant, candidate: ObservedVariant
841
+ ) -> list[ComparisonCheck]:
842
+ """Hold both executions to the container the Campaign pinned.
843
+
844
+ Decisions document 0007 R5. This used to be a warning, because the only
845
+ evidence of what ran was the reference the Campaign had asked for: the
846
+ pinned Verifiers build records no resolved digest, so "the image is the one
847
+ we asked for" was the strongest true statement available. It is a check now
848
+ because the Campaign pins a manifest digest per platform and every run asks
849
+ the local daemon what it holds, so the claim has evidence under it. An
850
+ execution that cannot be tied to the pin is invalid rather than warned
851
+ about — a container nobody can name is not a weaker result.
852
+ """
853
+ runtime = campaign.subject.runtime
854
+ pinned = runtime.image_index_digest
855
+ matched = sorted(
856
+ observed.variant.value
857
+ for observed in (baseline, candidate)
858
+ if observed.configuration.runtime_image_index_digest == pinned
859
+ )
860
+ both = len(matched) == 2
861
+
862
+ platforms = sorted(
863
+ {observed.configuration.runtime_platform for observed in (baseline, candidate)}
864
+ )
865
+ pinned_platform = len(platforms) == 1 and platforms[0] in (
866
+ runtime.image_platform_digests
867
+ )
868
+ return [
869
+ _check(
870
+ "observed_runtime_image_pinned",
871
+ _PASSED if both else _FAILED,
872
+ (
873
+ "the daemon confirmed both variants ran the image content the "
874
+ "Campaign pinned"
875
+ if both
876
+ else "the container the daemon holds is not the one the Campaign pinned"
877
+ ),
878
+ ),
879
+ _check(
880
+ "observed_runtime_platform_pinned",
881
+ _PASSED if pinned_platform else _FAILED,
882
+ (
883
+ f"both variants were served on {platforms[0]}, a platform the "
884
+ "Campaign pins a manifest digest for"
885
+ if pinned_platform
886
+ else "the variants were served on platforms the Campaign does "
887
+ "not pin one manifest digest for"
888
+ ),
889
+ ),
890
+ ]
891
+
892
+
893
+ def weaker_claim_warnings(campaign: CampaignSpec) -> list[ComparisonCheck]:
894
+ """Record the one fact about a real run that cannot be independently pinned.
895
+
896
+ It is not a scientific failure and it may not be left unsaid. Presenting
897
+ "both variants named the same model" as "both variants ran the same model
898
+ build" would be making the stronger claim from the weaker evidence, which is
899
+ the whole thing a controlled comparison exists to stop. Decisions document
900
+ 0007 R5 accepts it for v0.1 and forbids suppressing it to obtain the word
901
+ "controlled".
902
+ """
903
+ if campaign.subject.model.revision is not None:
904
+ return []
905
+ return [
906
+ _check(
907
+ MODEL_REVISION_UNDISCOVERABLE,
908
+ _WARNING,
909
+ f"the provider publishes no revision for "
910
+ f"{campaign.subject.model.model_id}, so both variants are known "
911
+ "to have used the same model identifier and not the same "
912
+ "model build",
913
+ )
914
+ ]
915
+
916
+
917
+ # ---------------------------------------------------------------------------
918
+ # The join
919
+ # ---------------------------------------------------------------------------
920
+
921
+
922
+ def _pair_receipts(
923
+ *,
924
+ baseline_receipts: Sequence[EpisodeReceipt],
925
+ candidate_receipts: Sequence[EpisodeReceipt],
926
+ committed: Sequence[Digest],
927
+ ) -> tuple[list[PairedReceiptRow], ComparisonCheck]:
928
+ """Join the two sides task by task, in committed order.
929
+
930
+ A missing task and a duplicated task are both reported rather than raised,
931
+ because they are findings about the comparison and belong in its list of
932
+ checks beside every other finding.
933
+ """
934
+ left, left_faults = _by_task(baseline_receipts, committed, "baseline")
935
+ right, right_faults = _by_task(candidate_receipts, committed, "candidate")
936
+ faults = [*left_faults, *right_faults]
937
+ if faults:
938
+ return [], _check("paired_task_rewards", _FAILED, "; ".join(faults))
939
+
940
+ rows = [
941
+ PairedReceiptRow(
942
+ position=position,
943
+ task_hash=task_hash,
944
+ baseline_receipt_id=left[task_hash].id,
945
+ baseline_receipt_digest=digest_object(left[task_hash]),
946
+ candidate_receipt_id=right[task_hash].id,
947
+ candidate_receipt_digest=digest_object(right[task_hash]),
948
+ )
949
+ for position, task_hash in enumerate(committed)
950
+ ]
951
+ return rows, _check(
952
+ "paired_task_rewards",
953
+ _PASSED,
954
+ f"every one of the {len(rows)} committed tasks is scored exactly once "
955
+ "on each side",
956
+ )
957
+
958
+
959
+ def _by_task(
960
+ receipts: Sequence[EpisodeReceipt],
961
+ committed: Sequence[Digest],
962
+ label: str,
963
+ ) -> tuple[dict[Digest, EpisodeReceipt], list[str]]:
964
+ """Index one side's receipts by task and report what does not line up."""
965
+ by_task: dict[Digest, EpisodeReceipt] = {}
966
+ faults: list[str] = []
967
+ for receipt in receipts:
968
+ if receipt.task_hash in by_task:
969
+ faults.append(f"the {label} scored {receipt.task_hash} twice")
970
+ continue
971
+ by_task[receipt.task_hash] = receipt
972
+
973
+ missing = [value for value in committed if value not in by_task]
974
+ unexpected = sorted(set(by_task) - set(committed))
975
+ if missing:
976
+ faults.append(f"the {label} did not score {len(missing)} committed task(s)")
977
+ if unexpected:
978
+ faults.append(
979
+ f"the {label} scored {len(unexpected)} task(s) the Campaign does "
980
+ "not commit to"
981
+ )
982
+ return by_task, faults
983
+
984
+
985
+ # ---------------------------------------------------------------------------
986
+ # Placement in time
987
+ # ---------------------------------------------------------------------------
988
+
989
+
990
+ def _observe_schedule(
991
+ *,
992
+ schedule: VariantSchedule,
993
+ baseline: ChildProcessOutcome,
994
+ candidate: ChildProcessOutcome,
995
+ ) -> ScheduleObservation:
996
+ """Record how far apart the two children started and finished."""
997
+ started = (baseline.started_at, candidate.started_at)
998
+ finished = (baseline.finished_at, candidate.finished_at)
999
+ return ScheduleObservation(
1000
+ schedule=schedule,
1001
+ start_skew_seconds=abs((started[1] - started[0]).total_seconds()),
1002
+ completion_window_seconds=(max(finished) - min(started)).total_seconds(),
1003
+ overlapped=max(started) < min(finished),
1004
+ )
1005
+
1006
+
1007
+ # ---------------------------------------------------------------------------
1008
+ # Small shared pieces
1009
+ # ---------------------------------------------------------------------------
1010
+
1011
+
1012
+ def _status_for(checks: Sequence[ComparisonCheck]) -> ComparisonStatus:
1013
+ """Return the one status a list of checks supports. Spec section 7.9."""
1014
+ if any(check.status == _FAILED for check in checks):
1015
+ return ComparisonStatus.INVALID
1016
+ if any(check.status == _WARNING for check in checks):
1017
+ return ComparisonStatus.CONTROLLED_WITH_WARNINGS
1018
+ return ComparisonStatus.CONTROLLED
1019
+
1020
+
1021
+ def _check(identifier: str, status: str, detail: str) -> ComparisonCheck:
1022
+ """Build one check, validated as the model it is."""
1023
+ return ComparisonCheck.model_validate(
1024
+ {"id": identifier, "status": status, "detail": detail}
1025
+ )
1026
+
1027
+
1028
+ def _same(identifier: str, label: str, left: object, right: object) -> ComparisonCheck:
1029
+ """Report whether two sides agree, without printing what they hold.
1030
+
1031
+ The values themselves are deliberately not in the detail. A model
1032
+ identifier is harmless, a resolved configuration is not, and one rule for
1033
+ all of them is the rule that cannot leak.
1034
+ """
1035
+ agree = left == right
1036
+ return _check(
1037
+ identifier,
1038
+ _PASSED if agree else _FAILED,
1039
+ (
1040
+ f"the baseline and the candidate share one {label}"
1041
+ if agree
1042
+ else f"the baseline and the candidate do not share one {label}"
1043
+ ),
1044
+ )
1045
+
1046
+
1047
+ def _subject(manifest: ExperimentManifest) -> AgentSpec:
1048
+ """Return the manifest's subject agent, which its own validator requires."""
1049
+ subject = manifest.configuration.agents.get(SUBJECT_AGENT)
1050
+ if subject is None: # pragma: no cover - the model refuses to be built without one
1051
+ raise ValueError("an experiment configuration defines a subject agent")
1052
+ return subject
1053
+
1054
+
1055
+ def _harness_identity(subject: AgentSpec) -> tuple[str, str, bool]:
1056
+ """Return the part of a harness two variants must share.
1057
+
1058
+ Not the skill list: that is the one field the mutation contract permits to
1059
+ differ, and comparing it here would fail every correct comparison.
1060
+ """
1061
+ return (
1062
+ subject.harness.id,
1063
+ subject.harness.version,
1064
+ subject.harness.use_bundled_skill,
1065
+ )