setspec 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- setspec/__about__.py +3 -0
- setspec/__init__.py +114 -0
- setspec/artifacts.py +4 -0
- setspec/base.py +256 -0
- setspec/benchmark/v1.py +594 -0
- setspec/capability/v1.py +394 -0
- setspec/envelope.py +468 -0
- setspec/error/v1.py +4 -0
- setspec/errors.py +68 -0
- setspec/event/v1.py +4 -0
- setspec/goal/v1.py +390 -0
- setspec/goldens/.gitkeep +0 -0
- setspec/machine/v1.py +131 -0
- setspec/metrics.py +184 -0
- setspec/model/v1.py +169 -0
- setspec/provenance.py +59 -0
- setspec/py.typed +0 -0
- setspec/schemas/.gitkeep +0 -0
- setspec/serialization.py +330 -0
- setspec/vocabulary.py +205 -0
- setspec-0.2.0.dist-info/METADATA +160 -0
- setspec-0.2.0.dist-info/RECORD +24 -0
- setspec-0.2.0.dist-info/WHEEL +4 -0
- setspec-0.2.0.dist-info/licenses/LICENSE +201 -0
setspec/capability/v1.py
ADDED
|
@@ -0,0 +1,394 @@
|
|
|
1
|
+
"""Contract module — ``capability.evidence`` and ``benchmark.evidence_bundle`` v1.
|
|
2
|
+
|
|
3
|
+
Imports pydantic and :mod:`baseaicore`; performs no I/O. This is the suite's most load-bearing
|
|
4
|
+
cross-application contract — the entire FreeWeight → LoadCoach value proposition
|
|
5
|
+
(ADR-0022) — so
|
|
6
|
+
``CapabilityEvidenceFields`` reproduces
|
|
7
|
+
ADR-0022 §1's normative field
|
|
8
|
+
table verbatim rather than approximating it: every field name, type and meaning below has that
|
|
9
|
+
table as its direct source, not this module's own judgment.
|
|
10
|
+
|
|
11
|
+
**Status: draft (`1.0`).** See :mod:`setspec.model.v1` for what that means; here it means
|
|
12
|
+
specifically that this module predates FreeWeight actually aggregating evidence, so
|
|
13
|
+
[Phase 4](../../../docs/packages/setspec/development-plan.md) may still adjust a field that turns
|
|
14
|
+
out to be shaped wrong once real aggregation exists to shape it against.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from typing import Self
|
|
20
|
+
|
|
21
|
+
from baseaicore import CapabilityId
|
|
22
|
+
from baseaicore import ValidationError as SuiteValidationError
|
|
23
|
+
from pydantic import Field, model_validator
|
|
24
|
+
|
|
25
|
+
from setspec import vocabulary
|
|
26
|
+
from setspec.base import PayloadDefinition, WireSequence, payload_models
|
|
27
|
+
from setspec.model.v1 import ModelIdentityFields
|
|
28
|
+
from setspec.provenance import EnvironmentFields
|
|
29
|
+
from setspec.serialization import MeasurementField, TimestampField
|
|
30
|
+
|
|
31
|
+
__all__ = [
|
|
32
|
+
"CalibrationFields",
|
|
33
|
+
"CapabilityEvidenceFields",
|
|
34
|
+
"CapabilityEvidenceIn",
|
|
35
|
+
"CapabilityEvidenceOut",
|
|
36
|
+
"ContributingMetricFields",
|
|
37
|
+
"EvidenceBundleFields",
|
|
38
|
+
"EvidenceBundleIn",
|
|
39
|
+
"EvidenceBundleOut",
|
|
40
|
+
"JudgeSetFields",
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
_MINIMUM_CONFIDENCE = 0.05
|
|
44
|
+
_MAXIMUM_SCORE_OR_CONFIDENCE = 1.0
|
|
45
|
+
_GOAL_NAMESPACE_ROOT = "user"
|
|
46
|
+
_SCORE_METHOD_RUNGS: frozenset[str] = frozenset({"rule", "reference", "human", "judge"})
|
|
47
|
+
_METHOD_MIX_TOLERANCE = 1e-6
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class ContributingMetricFields(PayloadDefinition):
|
|
51
|
+
"""One benchmark metric's contribution to a capability score.
|
|
52
|
+
|
|
53
|
+
Attributes:
|
|
54
|
+
metric_key: The metric's identifier within its benchmark, e.g. ``"task_success"``.
|
|
55
|
+
weight: This metric's weight in the score it contributed to. Positive — a zero or
|
|
56
|
+
negative weight would mean the metric contributed nothing or inverted another's
|
|
57
|
+
effect, either of which belongs in how the score was computed, not in a record of
|
|
58
|
+
what happened.
|
|
59
|
+
sample_count: Supported samples this metric contributed, matching the same
|
|
60
|
+
excluded-is-not-counted rule as every other sample count in this package
|
|
61
|
+
(ADR-0016 §6).
|
|
62
|
+
"""
|
|
63
|
+
|
|
64
|
+
metric_key: str = Field(min_length=1)
|
|
65
|
+
weight: float = Field(gt=0.0)
|
|
66
|
+
sample_count: int = Field(ge=0)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class JudgeSetFields(PayloadDefinition):
|
|
70
|
+
"""The instrument that produced a judged score — a hard-separation input (ADR-0032 §4).
|
|
71
|
+
|
|
72
|
+
A different jury is a different instrument, and therefore a different measurement. This is
|
|
73
|
+
recorded rather than summarized because a consumer must be able to decide comparability
|
|
74
|
+
without asking the producer, exactly as it does for a benchmark version.
|
|
75
|
+
|
|
76
|
+
Attributes:
|
|
77
|
+
jurors: Canonical model IDs of the jury members, in the order they were polled. Two or
|
|
78
|
+
more is the documented default; one is permitted and loses the inter-juror agreement
|
|
79
|
+
that distinguishes bias from noise, which is why the count is visible here rather
|
|
80
|
+
than folded into a summary figure.
|
|
81
|
+
prompt_id: The judge prompt record's ID (ADR-0012 — a judge rubric is a prompt record).
|
|
82
|
+
prompt_version: That record's semantic version.
|
|
83
|
+
prompt_sha256: That record's canonical hash, so a consumer can separate on the prompt
|
|
84
|
+
without holding the prompt.
|
|
85
|
+
remote: Whether any juror ran outside the measuring machine. Locally-judged and
|
|
86
|
+
remotely-judged results are **separated**, never merged (ADR-0031 §4), so this is a
|
|
87
|
+
comparability input and not a footnote.
|
|
88
|
+
"""
|
|
89
|
+
|
|
90
|
+
jurors: WireSequence[str] = ()
|
|
91
|
+
prompt_id: str = Field(min_length=1)
|
|
92
|
+
prompt_version: str = Field(min_length=1)
|
|
93
|
+
prompt_sha256: str = Field(min_length=1)
|
|
94
|
+
remote: bool
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
class CalibrationFields(PayloadDefinition):
|
|
98
|
+
"""How closely the judge agreed with the person whose goal this is (ADR-0031 §3).
|
|
99
|
+
|
|
100
|
+
This is what separates an instrument from an opinion. Every field here describes the *judge's*
|
|
101
|
+
error against user-supplied ground truth, not the measured model's performance.
|
|
102
|
+
|
|
103
|
+
Attributes:
|
|
104
|
+
kappa_w: Quadratic-weighted Cohen's kappa between the user's grades and the jury median
|
|
105
|
+
over the holdout, weighted across judged criteria. Ordinal-aware and chance-corrected;
|
|
106
|
+
legitimately negative when the judge disagrees with the user worse than chance would.
|
|
107
|
+
rho: Spearman rank correlation — whether the judge ranks as the user ranks.
|
|
108
|
+
mae: Mean absolute error in scale points. Non-negative, and in the units the user thinks
|
|
109
|
+
in rather than a coefficient they must interpret.
|
|
110
|
+
bias: Mean signed error. Negative means the judge grades harsher than the user, positive
|
|
111
|
+
more generously. Unbounded here because its scale is the criterion's, which this
|
|
112
|
+
payload does not carry.
|
|
113
|
+
n_anchor: Graded samples embedded in the judge prompt as exemplars. May be zero: a rubric
|
|
114
|
+
may be calibrated without few-shot anchoring.
|
|
115
|
+
n_holdout: Graded samples the judge was **never shown**, which is the only honest basis
|
|
116
|
+
for :attr:`kappa_w`. At least one, because agreement measured over nothing is not a
|
|
117
|
+
measurement — and it travels with the coefficient everywhere precisely so a reader
|
|
118
|
+
cannot see ``kappa_w`` without seeing what it was computed over.
|
|
119
|
+
graded_by: Free text the user supplied identifying the grader. Never an account name or
|
|
120
|
+
an address harvested from the environment.
|
|
121
|
+
measured_at: When the calibration was measured. Ages like evidence: a rubric calibrated a
|
|
122
|
+
year ago against a jury that has since changed is stale in the same sense a benchmark
|
|
123
|
+
result is.
|
|
124
|
+
"""
|
|
125
|
+
|
|
126
|
+
kappa_w: float = Field(ge=-1.0, le=1.0)
|
|
127
|
+
rho: float = Field(ge=-1.0, le=1.0)
|
|
128
|
+
mae: float = Field(ge=0.0)
|
|
129
|
+
bias: float
|
|
130
|
+
n_anchor: int = Field(ge=0)
|
|
131
|
+
n_holdout: int = Field(ge=1)
|
|
132
|
+
graded_by: str = Field(min_length=1)
|
|
133
|
+
measured_at: TimestampField
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
class CapabilityEvidenceFields(PayloadDefinition):
|
|
137
|
+
"""Field definitions for ``capability.evidence``; use :data:`CapabilityEvidenceOut` /
|
|
138
|
+
:data:`CapabilityEvidenceIn`.
|
|
139
|
+
|
|
140
|
+
Attributes:
|
|
141
|
+
model: The measured weights (ADR-0024).
|
|
142
|
+
runtime_profile_hash: The profile the measurement was taken under (ADR-0023).
|
|
143
|
+
machine_fingerprint: Where it was measured.
|
|
144
|
+
capability_id: A term in the SetSpec vocabulary; checked against
|
|
145
|
+
:func:`setspec.vocabulary.validate_capability` using :attr:`vocabulary_version` for
|
|
146
|
+
the forward-compatibility exception — see :meth:`_check_capability_id`.
|
|
147
|
+
score: The capability score, ``0`` to ``1`` inclusive.
|
|
148
|
+
confidence: Computed by FreeWeight per
|
|
149
|
+
ADR-0017; floored at
|
|
150
|
+
``0.05`` by that formula's own clamp, never truly zero.
|
|
151
|
+
sample_count: Supported samples that produced ``score``.
|
|
152
|
+
excluded_count: Samples excluded, with the exclusion visible rather than folded silently
|
|
153
|
+
into a lower ``sample_count``.
|
|
154
|
+
dispersion: Coefficient of variation for a continuous metric, or disagreement rate for a
|
|
155
|
+
pass/fail one — ADR-0017 defines which applies from the metric's own kind, not from a
|
|
156
|
+
flag carried here.
|
|
157
|
+
measured_at: The **latest ``completed_at`` among the contributing runs** — what
|
|
158
|
+
``freshness_factor`` decays from. Never the aggregation time.
|
|
159
|
+
computed_at: When this aggregation ran. Provenance and the incremental-export filter
|
|
160
|
+
input; never a confidence input — recomputing evidence must not make it look fresher
|
|
161
|
+
(ADR-0022 §2), and
|
|
162
|
+
:meth:`_check_measured_before_computed` enforces the one shape that rule requires:
|
|
163
|
+
``measured_at`` cannot be later than the aggregation that used it.
|
|
164
|
+
policy_version: The confidence-policy version this evidence was computed under.
|
|
165
|
+
vocabulary_version: The capability-vocabulary version ``capability_id`` was validated
|
|
166
|
+
against when this evidence was produced.
|
|
167
|
+
benchmark_versions: Suite key to version — a hard-separation input.
|
|
168
|
+
dataset_hashes: A hard-separation input.
|
|
169
|
+
prompt_subset_hashes: Hash **per benchmark key, not per pack**
|
|
170
|
+
(ADR-0028) — a hard-separation
|
|
171
|
+
input.
|
|
172
|
+
contributing_metrics: Which benchmark metrics fed this score, with what weight and how
|
|
173
|
+
many samples.
|
|
174
|
+
source_run_ids: Producer-local run IDs. A consumer stores these opaquely and never
|
|
175
|
+
resolves them.
|
|
176
|
+
environment: Provider kind and version, GPU driver, CUDA, OS version at measurement.
|
|
177
|
+
judge_validity_factor: The sixth confidence factor (ADR-0032 §2). **``1.0`` for every
|
|
178
|
+
measurement scored at ladder rungs 1–4**, which is every native and external
|
|
179
|
+
benchmark in the suite — so this field changed no existing number when it was added.
|
|
180
|
+
Below ``1.0`` only for a user-defined goal's judged criteria, in proportion to the
|
|
181
|
+
judge's measured agreement with the user and shrunk toward zero when the holdout is
|
|
182
|
+
small. Already multiplied into :attr:`confidence`; carried separately so a consumer
|
|
183
|
+
can *see* it without recomputing the formula.
|
|
184
|
+
goal_hash: The measurement-defining hash of the goal that produced this record, when one
|
|
185
|
+
did. A hard-separation input: a different rubric is a different measurement, exactly
|
|
186
|
+
as a different benchmark version is (ADR-0032 §4).
|
|
187
|
+
goal_pack_version: The goal pack's semantic version. Provenance; :attr:`goal_hash` is
|
|
188
|
+
what separates.
|
|
189
|
+
score_method_mix: Fraction of scored weight by ladder rung, e.g.
|
|
190
|
+
``{"rule": 0.6, "judge": 0.4}``. Keys are drawn from ``rule``, ``reference``,
|
|
191
|
+
``human`` and ``judge``; the values sum to ``1``. A ``0.82`` that is 80 % rules is a
|
|
192
|
+
different kind of number from a ``0.82`` that is 80 % judgement, and a consumer that
|
|
193
|
+
cannot tell them apart will eventually present them as the same thing.
|
|
194
|
+
judge_set: The jury that produced any judged portion — a hard-separation input.
|
|
195
|
+
calibration: The judge's measured agreement with the user. Present whenever a judged
|
|
196
|
+
criterion contributed; absent for a goal scored entirely by rules.
|
|
197
|
+
uncalibrated: Always ``False`` on a record that exists, and refused as ``True`` by
|
|
198
|
+
:meth:`_check_goal_fields_cohere`. A goal below its calibration gate emits **no
|
|
199
|
+
record at all** rather than a discounted one (ADR-0032 §3), so a ``True`` here means a
|
|
200
|
+
producer bug — one worth catching on the wire, where it is one field, rather than in a
|
|
201
|
+
routing decision months later, where it is a mystery.
|
|
202
|
+
"""
|
|
203
|
+
|
|
204
|
+
model: ModelIdentityFields
|
|
205
|
+
runtime_profile_hash: str = Field(min_length=1)
|
|
206
|
+
machine_fingerprint: str = Field(min_length=1)
|
|
207
|
+
capability_id: str = Field(min_length=1)
|
|
208
|
+
score: float = Field(ge=0.0, le=_MAXIMUM_SCORE_OR_CONFIDENCE)
|
|
209
|
+
confidence: float = Field(ge=_MINIMUM_CONFIDENCE, le=_MAXIMUM_SCORE_OR_CONFIDENCE)
|
|
210
|
+
sample_count: int = Field(ge=0)
|
|
211
|
+
excluded_count: int = Field(ge=0)
|
|
212
|
+
dispersion: MeasurementField
|
|
213
|
+
measured_at: TimestampField
|
|
214
|
+
computed_at: TimestampField
|
|
215
|
+
policy_version: str = Field(min_length=1)
|
|
216
|
+
vocabulary_version: str = Field(min_length=1)
|
|
217
|
+
benchmark_versions: dict[str, str] = Field(default_factory=dict)
|
|
218
|
+
dataset_hashes: dict[str, str] = Field(default_factory=dict)
|
|
219
|
+
prompt_subset_hashes: dict[str, str] = Field(default_factory=dict)
|
|
220
|
+
contributing_metrics: WireSequence[ContributingMetricFields] = ()
|
|
221
|
+
source_run_ids: WireSequence[str] = ()
|
|
222
|
+
environment: EnvironmentFields
|
|
223
|
+
# Goal-sourced group (ADR-0032 §5). Every field is optional and absent on a non-goal record,
|
|
224
|
+
# which is what makes this a minor schema change rather than a major one.
|
|
225
|
+
judge_validity_factor: float = Field(
|
|
226
|
+
default=1.0, ge=_MINIMUM_CONFIDENCE, le=_MAXIMUM_SCORE_OR_CONFIDENCE
|
|
227
|
+
)
|
|
228
|
+
goal_hash: str | None = Field(default=None, min_length=1)
|
|
229
|
+
goal_pack_version: str | None = Field(default=None, min_length=1)
|
|
230
|
+
score_method_mix: dict[str, float] | None = None
|
|
231
|
+
judge_set: JudgeSetFields | None = None
|
|
232
|
+
calibration: CalibrationFields | None = None
|
|
233
|
+
uncalibrated: bool = False
|
|
234
|
+
|
|
235
|
+
@model_validator(mode="after")
|
|
236
|
+
def _check_capability_id(self) -> Self:
|
|
237
|
+
"""Validate ``capability_id`` against the vocabulary at ``vocabulary_version``.
|
|
238
|
+
|
|
239
|
+
Raises:
|
|
240
|
+
ValueError: If ``capability_id`` is syntactically invalid, or its root is unrecognized
|
|
241
|
+
and :attr:`vocabulary_version` does not prove forward compatibility.
|
|
242
|
+
"""
|
|
243
|
+
try:
|
|
244
|
+
vocabulary.validate_capability(
|
|
245
|
+
self.capability_id, vocabulary_version=self.vocabulary_version
|
|
246
|
+
)
|
|
247
|
+
except SuiteValidationError as exc:
|
|
248
|
+
raise ValueError(str(exc)) from exc
|
|
249
|
+
return self
|
|
250
|
+
|
|
251
|
+
@model_validator(mode="after")
|
|
252
|
+
def _check_measured_before_computed(self) -> Self:
|
|
253
|
+
"""Require ``measured_at`` not to be later than ``computed_at``.
|
|
254
|
+
|
|
255
|
+
Raises:
|
|
256
|
+
ValueError: If ``measured_at`` follows ``computed_at`` — incoherent, since
|
|
257
|
+
``computed_at`` is when the aggregation ran and cannot precede what it aggregated
|
|
258
|
+
(ADR-0022 §2).
|
|
259
|
+
"""
|
|
260
|
+
if self.measured_at > self.computed_at:
|
|
261
|
+
raise ValueError(
|
|
262
|
+
f"measured_at ({self.measured_at.isoformat()}) is later than computed_at "
|
|
263
|
+
f"({self.computed_at.isoformat()}). computed_at is when this aggregation ran and "
|
|
264
|
+
"cannot precede the measurement it aggregated — freshness decays from "
|
|
265
|
+
"measured_at, never computed_at, precisely so this cannot be gamed by "
|
|
266
|
+
"recomputing (ADR-0022 §2)."
|
|
267
|
+
)
|
|
268
|
+
return self
|
|
269
|
+
|
|
270
|
+
@model_validator(mode="after")
|
|
271
|
+
def _check_goal_fields_cohere(self) -> Self:
|
|
272
|
+
"""Enforce the goal-sourced group's internal rules (ADR-0032 §1–§5).
|
|
273
|
+
|
|
274
|
+
Five rules, each closing a way a subjective score could quietly acquire authority it has
|
|
275
|
+
not earned:
|
|
276
|
+
|
|
277
|
+
1. ``uncalibrated`` is never ``True``. The gate withholds the record entirely rather
|
|
278
|
+
than emitting a discounted one, so a record that says it is uncalibrated is one the
|
|
279
|
+
producer should not have written.
|
|
280
|
+
2. A ``user.*`` capability carries a ``goal_hash``. Without it the record cannot be
|
|
281
|
+
separated from a differently-defined goal of the same name.
|
|
282
|
+
3. A record with no ``goal_hash`` has ``judge_validity_factor`` exactly ``1.0``. Nothing
|
|
283
|
+
but a goal can discount validity, and a discount with no goal attached is unexplainable.
|
|
284
|
+
4. A discounted ``judge_validity_factor`` carries the ``calibration`` it came from. The
|
|
285
|
+
number is derived from measured agreement; without that agreement it is an assertion.
|
|
286
|
+
5. A ``calibration`` carries the ``judge_set`` it measured. Agreement is a property of a
|
|
287
|
+
particular jury, and a jury change separates results — so agreement without the jury's
|
|
288
|
+
identity cannot be applied.
|
|
289
|
+
|
|
290
|
+
Raises:
|
|
291
|
+
ValueError: If any of the five rules is broken, naming which and why.
|
|
292
|
+
"""
|
|
293
|
+
if self.uncalibrated:
|
|
294
|
+
raise ValueError(
|
|
295
|
+
"uncalibrated must be False on an emitted record. A goal below its calibration "
|
|
296
|
+
"gate emits no capability.evidence at all — not a discounted record "
|
|
297
|
+
"(ADR-0032 §3). A True here means the producer wrote a record the gate should "
|
|
298
|
+
"have withheld."
|
|
299
|
+
)
|
|
300
|
+
root = CapabilityId(self.capability_id).root
|
|
301
|
+
if root == _GOAL_NAMESPACE_ROOT and self.goal_hash is None:
|
|
302
|
+
raise ValueError(
|
|
303
|
+
f"capability_id {self.capability_id!r} is in the reserved 'user' namespace but "
|
|
304
|
+
"carries no goal_hash. A goal's identity is its hash, not its slug: two people's "
|
|
305
|
+
"'user.house_voice' are different measurements, and without the hash a consumer "
|
|
306
|
+
"cannot separate them (ADR-0032 §4)."
|
|
307
|
+
)
|
|
308
|
+
if self.goal_hash is None and self.judge_validity_factor != 1.0:
|
|
309
|
+
raise ValueError(
|
|
310
|
+
f"judge_validity_factor is {self.judge_validity_factor} on a record with no "
|
|
311
|
+
"goal_hash. Only a user-defined goal's judged criteria can reduce validity; "
|
|
312
|
+
"every rung 1-4 measurement is exactly 1.0 (ADR-0032 §2)."
|
|
313
|
+
)
|
|
314
|
+
if self.judge_validity_factor < 1.0 and self.calibration is None:
|
|
315
|
+
raise ValueError(
|
|
316
|
+
f"judge_validity_factor is {self.judge_validity_factor} but no calibration is "
|
|
317
|
+
"present. The factor is derived from measured judge-user agreement; without the "
|
|
318
|
+
"calibration it came from it is an assertion, and a consumer cannot audit it."
|
|
319
|
+
)
|
|
320
|
+
if self.calibration is not None and self.judge_set is None:
|
|
321
|
+
raise ValueError(
|
|
322
|
+
"calibration is present but judge_set is not. Agreement is a property of a "
|
|
323
|
+
"particular jury — change the jury and the agreement no longer applies — so the "
|
|
324
|
+
"jury's identity travels with it (ADR-0032 §4)."
|
|
325
|
+
)
|
|
326
|
+
return self
|
|
327
|
+
|
|
328
|
+
@model_validator(mode="after")
|
|
329
|
+
def _check_score_method_mix(self) -> Self:
|
|
330
|
+
"""Require ``score_method_mix`` to name known ladder rungs and to sum to ``1``.
|
|
331
|
+
|
|
332
|
+
Raises:
|
|
333
|
+
ValueError: If a key is not one of ``rule``, ``reference``, ``human`` or ``judge``, if
|
|
334
|
+
a fraction falls outside ``[0, 1]``, or if the fractions do not sum to ``1``. A
|
|
335
|
+
mix that does not sum to one describes scored weight that went somewhere
|
|
336
|
+
unaccounted for, which makes every share in it wrong rather than merely
|
|
337
|
+
incomplete.
|
|
338
|
+
"""
|
|
339
|
+
if self.score_method_mix is None:
|
|
340
|
+
return self
|
|
341
|
+
unknown = sorted(set(self.score_method_mix) - _SCORE_METHOD_RUNGS)
|
|
342
|
+
if unknown:
|
|
343
|
+
raise ValueError(
|
|
344
|
+
f"score_method_mix names unknown scoring rungs {unknown}. The ladder's rungs are "
|
|
345
|
+
f"{sorted(_SCORE_METHOD_RUNGS)} (benchmark catalog §1)."
|
|
346
|
+
)
|
|
347
|
+
out_of_range = sorted(k for k, v in self.score_method_mix.items() if not 0.0 <= v <= 1.0)
|
|
348
|
+
if out_of_range:
|
|
349
|
+
raise ValueError(
|
|
350
|
+
f"score_method_mix values must be fractions in [0, 1]; {out_of_range} are not."
|
|
351
|
+
)
|
|
352
|
+
total = sum(self.score_method_mix.values())
|
|
353
|
+
if abs(total - 1.0) > _METHOD_MIX_TOLERANCE:
|
|
354
|
+
raise ValueError(
|
|
355
|
+
f"score_method_mix sums to {total}, not 1. It describes how the scored weight was "
|
|
356
|
+
"divided, so weight that is unaccounted for makes every share in the mix wrong, "
|
|
357
|
+
"not merely the total."
|
|
358
|
+
)
|
|
359
|
+
return self
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
CapabilityEvidenceOut, CapabilityEvidenceIn = payload_models(CapabilityEvidenceFields)
|
|
363
|
+
"""The ``capability.evidence`` payload pair: ``Out`` for writers, ``In`` for readers."""
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
class EvidenceBundleFields(PayloadDefinition):
|
|
367
|
+
"""Field definitions for ``benchmark.evidence_bundle``; use :data:`EvidenceBundleOut` /
|
|
368
|
+
:data:`EvidenceBundleIn`.
|
|
369
|
+
|
|
370
|
+
The FreeWeight → LoadCoach payload: many :class:`CapabilityEvidenceFields`, plus the one flag
|
|
371
|
+
that makes incremental import possible
|
|
372
|
+
(ADR-0022 §5).
|
|
373
|
+
``generated_at`` is **not** repeated here: it lives on the enclosing
|
|
374
|
+
:class:`~setspec.envelope.SchemaEnvelope`, and a client stores *that* value to send back as its
|
|
375
|
+
next ``?since=`` — duplicating it on the payload would create two timestamps that could
|
|
376
|
+
disagree about when this bundle was produced.
|
|
377
|
+
|
|
378
|
+
Attributes:
|
|
379
|
+
source_id: Which FreeWeight instance produced this bundle — part of a consumer's
|
|
380
|
+
uniqueness key alongside ``canonical_id``, ``runtime_profile_hash``,
|
|
381
|
+
``machine_fingerprint``, ``capability_id`` and ``policy_version``.
|
|
382
|
+
complete: ``True`` only for a full export. Only a complete bundle lets a consumer infer
|
|
383
|
+
removal: evidence present locally for this ``source_id`` and absent from a complete
|
|
384
|
+
bundle is marked superseded — never deleted, and never inferred from a partial one.
|
|
385
|
+
evidence: The evidence records themselves.
|
|
386
|
+
"""
|
|
387
|
+
|
|
388
|
+
source_id: str = Field(min_length=1)
|
|
389
|
+
complete: bool
|
|
390
|
+
evidence: WireSequence[CapabilityEvidenceFields] = ()
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
EvidenceBundleOut, EvidenceBundleIn = payload_models(EvidenceBundleFields)
|
|
394
|
+
"""The ``benchmark.evidence_bundle`` payload pair: ``Out`` for writers, ``In`` for readers."""
|