setspec 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,394 @@
1
+ """Contract module — ``capability.evidence`` and ``benchmark.evidence_bundle`` v1.
2
+
3
+ Imports pydantic and :mod:`baseaicore`; performs no I/O. This is the suite's most load-bearing
4
+ cross-application contract — the entire FreeWeight → LoadCoach value proposition
5
+ (ADR-0022) — so
6
+ ``CapabilityEvidenceFields`` reproduces
7
+ ADR-0022 §1's normative field
8
+ table verbatim rather than approximating it: every field name, type and meaning below has that
9
+ table as its direct source, not this module's own judgment.
10
+
11
+ **Status: draft (`1.0`).** See :mod:`setspec.model.v1` for what that means; here it means
12
+ specifically that this module predates FreeWeight actually aggregating evidence, so
13
+ [Phase 4](../../../docs/packages/setspec/development-plan.md) may still adjust a field that turns
14
+ out to be shaped wrong once real aggregation exists to shape it against.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from typing import Self
20
+
21
+ from baseaicore import CapabilityId
22
+ from baseaicore import ValidationError as SuiteValidationError
23
+ from pydantic import Field, model_validator
24
+
25
+ from setspec import vocabulary
26
+ from setspec.base import PayloadDefinition, WireSequence, payload_models
27
+ from setspec.model.v1 import ModelIdentityFields
28
+ from setspec.provenance import EnvironmentFields
29
+ from setspec.serialization import MeasurementField, TimestampField
30
+
31
+ __all__ = [
32
+ "CalibrationFields",
33
+ "CapabilityEvidenceFields",
34
+ "CapabilityEvidenceIn",
35
+ "CapabilityEvidenceOut",
36
+ "ContributingMetricFields",
37
+ "EvidenceBundleFields",
38
+ "EvidenceBundleIn",
39
+ "EvidenceBundleOut",
40
+ "JudgeSetFields",
41
+ ]
42
+
43
+ _MINIMUM_CONFIDENCE = 0.05
44
+ _MAXIMUM_SCORE_OR_CONFIDENCE = 1.0
45
+ _GOAL_NAMESPACE_ROOT = "user"
46
+ _SCORE_METHOD_RUNGS: frozenset[str] = frozenset({"rule", "reference", "human", "judge"})
47
+ _METHOD_MIX_TOLERANCE = 1e-6
48
+
49
+
50
+ class ContributingMetricFields(PayloadDefinition):
51
+ """One benchmark metric's contribution to a capability score.
52
+
53
+ Attributes:
54
+ metric_key: The metric's identifier within its benchmark, e.g. ``"task_success"``.
55
+ weight: This metric's weight in the score it contributed to. Positive — a zero or
56
+ negative weight would mean the metric contributed nothing or inverted another's
57
+ effect, either of which belongs in how the score was computed, not in a record of
58
+ what happened.
59
+ sample_count: Supported samples this metric contributed, matching the same
60
+ excluded-is-not-counted rule as every other sample count in this package
61
+ (ADR-0016 §6).
62
+ """
63
+
64
+ metric_key: str = Field(min_length=1)
65
+ weight: float = Field(gt=0.0)
66
+ sample_count: int = Field(ge=0)
67
+
68
+
69
+ class JudgeSetFields(PayloadDefinition):
70
+ """The instrument that produced a judged score — a hard-separation input (ADR-0032 §4).
71
+
72
+ A different jury is a different instrument, and therefore a different measurement. This is
73
+ recorded rather than summarized because a consumer must be able to decide comparability
74
+ without asking the producer, exactly as it does for a benchmark version.
75
+
76
+ Attributes:
77
+ jurors: Canonical model IDs of the jury members, in the order they were polled. Two or
78
+ more is the documented default; one is permitted and loses the inter-juror agreement
79
+ that distinguishes bias from noise, which is why the count is visible here rather
80
+ than folded into a summary figure.
81
+ prompt_id: The judge prompt record's ID (ADR-0012 — a judge rubric is a prompt record).
82
+ prompt_version: That record's semantic version.
83
+ prompt_sha256: That record's canonical hash, so a consumer can separate on the prompt
84
+ without holding the prompt.
85
+ remote: Whether any juror ran outside the measuring machine. Locally-judged and
86
+ remotely-judged results are **separated**, never merged (ADR-0031 §4), so this is a
87
+ comparability input and not a footnote.
88
+ """
89
+
90
+ jurors: WireSequence[str] = ()
91
+ prompt_id: str = Field(min_length=1)
92
+ prompt_version: str = Field(min_length=1)
93
+ prompt_sha256: str = Field(min_length=1)
94
+ remote: bool
95
+
96
+
97
+ class CalibrationFields(PayloadDefinition):
98
+ """How closely the judge agreed with the person whose goal this is (ADR-0031 §3).
99
+
100
+ This is what separates an instrument from an opinion. Every field here describes the *judge's*
101
+ error against user-supplied ground truth, not the measured model's performance.
102
+
103
+ Attributes:
104
+ kappa_w: Quadratic-weighted Cohen's kappa between the user's grades and the jury median
105
+ over the holdout, weighted across judged criteria. Ordinal-aware and chance-corrected;
106
+ legitimately negative when the judge disagrees with the user worse than chance would.
107
+ rho: Spearman rank correlation — whether the judge ranks as the user ranks.
108
+ mae: Mean absolute error in scale points. Non-negative, and in the units the user thinks
109
+ in rather than a coefficient they must interpret.
110
+ bias: Mean signed error. Negative means the judge grades harsher than the user, positive
111
+ more generously. Unbounded here because its scale is the criterion's, which this
112
+ payload does not carry.
113
+ n_anchor: Graded samples embedded in the judge prompt as exemplars. May be zero: a rubric
114
+ may be calibrated without few-shot anchoring.
115
+ n_holdout: Graded samples the judge was **never shown**, which is the only honest basis
116
+ for :attr:`kappa_w`. At least one, because agreement measured over nothing is not a
117
+ measurement — and it travels with the coefficient everywhere precisely so a reader
118
+ cannot see ``kappa_w`` without seeing what it was computed over.
119
+ graded_by: Free text the user supplied identifying the grader. Never an account name or
120
+ an address harvested from the environment.
121
+ measured_at: When the calibration was measured. Ages like evidence: a rubric calibrated a
122
+ year ago against a jury that has since changed is stale in the same sense a benchmark
123
+ result is.
124
+ """
125
+
126
+ kappa_w: float = Field(ge=-1.0, le=1.0)
127
+ rho: float = Field(ge=-1.0, le=1.0)
128
+ mae: float = Field(ge=0.0)
129
+ bias: float
130
+ n_anchor: int = Field(ge=0)
131
+ n_holdout: int = Field(ge=1)
132
+ graded_by: str = Field(min_length=1)
133
+ measured_at: TimestampField
134
+
135
+
136
+ class CapabilityEvidenceFields(PayloadDefinition):
137
+ """Field definitions for ``capability.evidence``; use :data:`CapabilityEvidenceOut` /
138
+ :data:`CapabilityEvidenceIn`.
139
+
140
+ Attributes:
141
+ model: The measured weights (ADR-0024).
142
+ runtime_profile_hash: The profile the measurement was taken under (ADR-0023).
143
+ machine_fingerprint: Where it was measured.
144
+ capability_id: A term in the SetSpec vocabulary; checked against
145
+ :func:`setspec.vocabulary.validate_capability` using :attr:`vocabulary_version` for
146
+ the forward-compatibility exception — see :meth:`_check_capability_id`.
147
+ score: The capability score, ``0`` to ``1`` inclusive.
148
+ confidence: Computed by FreeWeight per
149
+ ADR-0017; floored at
150
+ ``0.05`` by that formula's own clamp, never truly zero.
151
+ sample_count: Supported samples that produced ``score``.
152
+ excluded_count: Samples excluded, with the exclusion visible rather than folded silently
153
+ into a lower ``sample_count``.
154
+ dispersion: Coefficient of variation for a continuous metric, or disagreement rate for a
155
+ pass/fail one — ADR-0017 defines which applies from the metric's own kind, not from a
156
+ flag carried here.
157
+ measured_at: The **latest ``completed_at`` among the contributing runs** — what
158
+ ``freshness_factor`` decays from. Never the aggregation time.
159
+ computed_at: When this aggregation ran. Provenance and the incremental-export filter
160
+ input; never a confidence input — recomputing evidence must not make it look fresher
161
+ (ADR-0022 §2), and
162
+ :meth:`_check_measured_before_computed` enforces the one shape that rule requires:
163
+ ``measured_at`` cannot be later than the aggregation that used it.
164
+ policy_version: The confidence-policy version this evidence was computed under.
165
+ vocabulary_version: The capability-vocabulary version ``capability_id`` was validated
166
+ against when this evidence was produced.
167
+ benchmark_versions: Suite key to version — a hard-separation input.
168
+ dataset_hashes: A hard-separation input.
169
+ prompt_subset_hashes: Hash **per benchmark key, not per pack**
170
+ (ADR-0028) — a hard-separation
171
+ input.
172
+ contributing_metrics: Which benchmark metrics fed this score, with what weight and how
173
+ many samples.
174
+ source_run_ids: Producer-local run IDs. A consumer stores these opaquely and never
175
+ resolves them.
176
+ environment: Provider kind and version, GPU driver, CUDA, OS version at measurement.
177
+ judge_validity_factor: The sixth confidence factor (ADR-0032 §2). **``1.0`` for every
178
+ measurement scored at ladder rungs 1–4**, which is every native and external
179
+ benchmark in the suite — so this field changed no existing number when it was added.
180
+ Below ``1.0`` only for a user-defined goal's judged criteria, in proportion to the
181
+ judge's measured agreement with the user and shrunk toward zero when the holdout is
182
+ small. Already multiplied into :attr:`confidence`; carried separately so a consumer
183
+ can *see* it without recomputing the formula.
184
+ goal_hash: The measurement-defining hash of the goal that produced this record, when one
185
+ did. A hard-separation input: a different rubric is a different measurement, exactly
186
+ as a different benchmark version is (ADR-0032 §4).
187
+ goal_pack_version: The goal pack's semantic version. Provenance; :attr:`goal_hash` is
188
+ what separates.
189
+ score_method_mix: Fraction of scored weight by ladder rung, e.g.
190
+ ``{"rule": 0.6, "judge": 0.4}``. Keys are drawn from ``rule``, ``reference``,
191
+ ``human`` and ``judge``; the values sum to ``1``. A ``0.82`` that is 80 % rules is a
192
+ different kind of number from a ``0.82`` that is 80 % judgement, and a consumer that
193
+ cannot tell them apart will eventually present them as the same thing.
194
+ judge_set: The jury that produced any judged portion — a hard-separation input.
195
+ calibration: The judge's measured agreement with the user. Present whenever a judged
196
+ criterion contributed; absent for a goal scored entirely by rules.
197
+ uncalibrated: Always ``False`` on a record that exists, and refused as ``True`` by
198
+ :meth:`_check_goal_fields_cohere`. A goal below its calibration gate emits **no
199
+ record at all** rather than a discounted one (ADR-0032 §3), so a ``True`` here means a
200
+ producer bug — one worth catching on the wire, where it is one field, rather than in a
201
+ routing decision months later, where it is a mystery.
202
+ """
203
+
204
+ model: ModelIdentityFields
205
+ runtime_profile_hash: str = Field(min_length=1)
206
+ machine_fingerprint: str = Field(min_length=1)
207
+ capability_id: str = Field(min_length=1)
208
+ score: float = Field(ge=0.0, le=_MAXIMUM_SCORE_OR_CONFIDENCE)
209
+ confidence: float = Field(ge=_MINIMUM_CONFIDENCE, le=_MAXIMUM_SCORE_OR_CONFIDENCE)
210
+ sample_count: int = Field(ge=0)
211
+ excluded_count: int = Field(ge=0)
212
+ dispersion: MeasurementField
213
+ measured_at: TimestampField
214
+ computed_at: TimestampField
215
+ policy_version: str = Field(min_length=1)
216
+ vocabulary_version: str = Field(min_length=1)
217
+ benchmark_versions: dict[str, str] = Field(default_factory=dict)
218
+ dataset_hashes: dict[str, str] = Field(default_factory=dict)
219
+ prompt_subset_hashes: dict[str, str] = Field(default_factory=dict)
220
+ contributing_metrics: WireSequence[ContributingMetricFields] = ()
221
+ source_run_ids: WireSequence[str] = ()
222
+ environment: EnvironmentFields
223
+ # Goal-sourced group (ADR-0032 §5). Every field is optional and absent on a non-goal record,
224
+ # which is what makes this a minor schema change rather than a major one.
225
+ judge_validity_factor: float = Field(
226
+ default=1.0, ge=_MINIMUM_CONFIDENCE, le=_MAXIMUM_SCORE_OR_CONFIDENCE
227
+ )
228
+ goal_hash: str | None = Field(default=None, min_length=1)
229
+ goal_pack_version: str | None = Field(default=None, min_length=1)
230
+ score_method_mix: dict[str, float] | None = None
231
+ judge_set: JudgeSetFields | None = None
232
+ calibration: CalibrationFields | None = None
233
+ uncalibrated: bool = False
234
+
235
+ @model_validator(mode="after")
236
+ def _check_capability_id(self) -> Self:
237
+ """Validate ``capability_id`` against the vocabulary at ``vocabulary_version``.
238
+
239
+ Raises:
240
+ ValueError: If ``capability_id`` is syntactically invalid, or its root is unrecognized
241
+ and :attr:`vocabulary_version` does not prove forward compatibility.
242
+ """
243
+ try:
244
+ vocabulary.validate_capability(
245
+ self.capability_id, vocabulary_version=self.vocabulary_version
246
+ )
247
+ except SuiteValidationError as exc:
248
+ raise ValueError(str(exc)) from exc
249
+ return self
250
+
251
+ @model_validator(mode="after")
252
+ def _check_measured_before_computed(self) -> Self:
253
+ """Require ``measured_at`` not to be later than ``computed_at``.
254
+
255
+ Raises:
256
+ ValueError: If ``measured_at`` follows ``computed_at`` — incoherent, since
257
+ ``computed_at`` is when the aggregation ran and cannot precede what it aggregated
258
+ (ADR-0022 §2).
259
+ """
260
+ if self.measured_at > self.computed_at:
261
+ raise ValueError(
262
+ f"measured_at ({self.measured_at.isoformat()}) is later than computed_at "
263
+ f"({self.computed_at.isoformat()}). computed_at is when this aggregation ran and "
264
+ "cannot precede the measurement it aggregated — freshness decays from "
265
+ "measured_at, never computed_at, precisely so this cannot be gamed by "
266
+ "recomputing (ADR-0022 §2)."
267
+ )
268
+ return self
269
+
270
+ @model_validator(mode="after")
271
+ def _check_goal_fields_cohere(self) -> Self:
272
+ """Enforce the goal-sourced group's internal rules (ADR-0032 §1–§5).
273
+
274
+ Five rules, each closing a way a subjective score could quietly acquire authority it has
275
+ not earned:
276
+
277
+ 1. ``uncalibrated`` is never ``True``. The gate withholds the record entirely rather
278
+ than emitting a discounted one, so a record that says it is uncalibrated is one the
279
+ producer should not have written.
280
+ 2. A ``user.*`` capability carries a ``goal_hash``. Without it the record cannot be
281
+ separated from a differently-defined goal of the same name.
282
+ 3. A record with no ``goal_hash`` has ``judge_validity_factor`` exactly ``1.0``. Nothing
283
+ but a goal can discount validity, and a discount with no goal attached is unexplainable.
284
+ 4. A discounted ``judge_validity_factor`` carries the ``calibration`` it came from. The
285
+ number is derived from measured agreement; without that agreement it is an assertion.
286
+ 5. A ``calibration`` carries the ``judge_set`` it measured. Agreement is a property of a
287
+ particular jury, and a jury change separates results — so agreement without the jury's
288
+ identity cannot be applied.
289
+
290
+ Raises:
291
+ ValueError: If any of the five rules is broken, naming which and why.
292
+ """
293
+ if self.uncalibrated:
294
+ raise ValueError(
295
+ "uncalibrated must be False on an emitted record. A goal below its calibration "
296
+ "gate emits no capability.evidence at all — not a discounted record "
297
+ "(ADR-0032 §3). A True here means the producer wrote a record the gate should "
298
+ "have withheld."
299
+ )
300
+ root = CapabilityId(self.capability_id).root
301
+ if root == _GOAL_NAMESPACE_ROOT and self.goal_hash is None:
302
+ raise ValueError(
303
+ f"capability_id {self.capability_id!r} is in the reserved 'user' namespace but "
304
+ "carries no goal_hash. A goal's identity is its hash, not its slug: two people's "
305
+ "'user.house_voice' are different measurements, and without the hash a consumer "
306
+ "cannot separate them (ADR-0032 §4)."
307
+ )
308
+ if self.goal_hash is None and self.judge_validity_factor != 1.0:
309
+ raise ValueError(
310
+ f"judge_validity_factor is {self.judge_validity_factor} on a record with no "
311
+ "goal_hash. Only a user-defined goal's judged criteria can reduce validity; "
312
+ "every rung 1-4 measurement is exactly 1.0 (ADR-0032 §2)."
313
+ )
314
+ if self.judge_validity_factor < 1.0 and self.calibration is None:
315
+ raise ValueError(
316
+ f"judge_validity_factor is {self.judge_validity_factor} but no calibration is "
317
+ "present. The factor is derived from measured judge-user agreement; without the "
318
+ "calibration it came from it is an assertion, and a consumer cannot audit it."
319
+ )
320
+ if self.calibration is not None and self.judge_set is None:
321
+ raise ValueError(
322
+ "calibration is present but judge_set is not. Agreement is a property of a "
323
+ "particular jury — change the jury and the agreement no longer applies — so the "
324
+ "jury's identity travels with it (ADR-0032 §4)."
325
+ )
326
+ return self
327
+
328
+ @model_validator(mode="after")
329
+ def _check_score_method_mix(self) -> Self:
330
+ """Require ``score_method_mix`` to name known ladder rungs and to sum to ``1``.
331
+
332
+ Raises:
333
+ ValueError: If a key is not one of ``rule``, ``reference``, ``human`` or ``judge``, if
334
+ a fraction falls outside ``[0, 1]``, or if the fractions do not sum to ``1``. A
335
+ mix that does not sum to one describes scored weight that went somewhere
336
+ unaccounted for, which makes every share in it wrong rather than merely
337
+ incomplete.
338
+ """
339
+ if self.score_method_mix is None:
340
+ return self
341
+ unknown = sorted(set(self.score_method_mix) - _SCORE_METHOD_RUNGS)
342
+ if unknown:
343
+ raise ValueError(
344
+ f"score_method_mix names unknown scoring rungs {unknown}. The ladder's rungs are "
345
+ f"{sorted(_SCORE_METHOD_RUNGS)} (benchmark catalog §1)."
346
+ )
347
+ out_of_range = sorted(k for k, v in self.score_method_mix.items() if not 0.0 <= v <= 1.0)
348
+ if out_of_range:
349
+ raise ValueError(
350
+ f"score_method_mix values must be fractions in [0, 1]; {out_of_range} are not."
351
+ )
352
+ total = sum(self.score_method_mix.values())
353
+ if abs(total - 1.0) > _METHOD_MIX_TOLERANCE:
354
+ raise ValueError(
355
+ f"score_method_mix sums to {total}, not 1. It describes how the scored weight was "
356
+ "divided, so weight that is unaccounted for makes every share in the mix wrong, "
357
+ "not merely the total."
358
+ )
359
+ return self
360
+
361
+
362
+ CapabilityEvidenceOut, CapabilityEvidenceIn = payload_models(CapabilityEvidenceFields)
363
+ """The ``capability.evidence`` payload pair: ``Out`` for writers, ``In`` for readers."""
364
+
365
+
366
+ class EvidenceBundleFields(PayloadDefinition):
367
+ """Field definitions for ``benchmark.evidence_bundle``; use :data:`EvidenceBundleOut` /
368
+ :data:`EvidenceBundleIn`.
369
+
370
+ The FreeWeight → LoadCoach payload: many :class:`CapabilityEvidenceFields`, plus the one flag
371
+ that makes incremental import possible
372
+ (ADR-0022 §5).
373
+ ``generated_at`` is **not** repeated here: it lives on the enclosing
374
+ :class:`~setspec.envelope.SchemaEnvelope`, and a client stores *that* value to send back as its
375
+ next ``?since=`` — duplicating it on the payload would create two timestamps that could
376
+ disagree about when this bundle was produced.
377
+
378
+ Attributes:
379
+ source_id: Which FreeWeight instance produced this bundle — part of a consumer's
380
+ uniqueness key alongside ``canonical_id``, ``runtime_profile_hash``,
381
+ ``machine_fingerprint``, ``capability_id`` and ``policy_version``.
382
+ complete: ``True`` only for a full export. Only a complete bundle lets a consumer infer
383
+ removal: evidence present locally for this ``source_id`` and absent from a complete
384
+ bundle is marked superseded — never deleted, and never inferred from a partial one.
385
+ evidence: The evidence records themselves.
386
+ """
387
+
388
+ source_id: str = Field(min_length=1)
389
+ complete: bool
390
+ evidence: WireSequence[CapabilityEvidenceFields] = ()
391
+
392
+
393
+ EvidenceBundleOut, EvidenceBundleIn = payload_models(EvidenceBundleFields)
394
+ """The ``benchmark.evidence_bundle`` payload pair: ``Out`` for writers, ``In`` for readers."""