setspec 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
setspec/metrics.py ADDED
@@ -0,0 +1,184 @@
1
+ """Contract module — the measured value that every benchmark metric reduces to.
2
+
3
+ Imports pydantic and :mod:`baseaicore`; performs no I/O and computes nothing. SetSpec carries
4
+ statistics, it never produces them: the aggregation named here was performed by the producer, and
5
+ this model's job is to make the producer state what it did rather than hand over a bare number
6
+ (spec §3, non-goals).
7
+
8
+ The invariants below are the schema-level half of
9
+ ADR-0016 §6. That ADR requires unsupported
10
+ samples to be *excluded* from a statistic and the surviving sample count to be reported next to it;
11
+ a model that let a caller write ``value=0.0, sample_count=0`` would make the rule advisory. Here it
12
+ is structural — the two fields cannot disagree, so a metric that was never measured cannot be
13
+ serialized as one that measured zero.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from enum import StrEnum
19
+ from typing import Self
20
+
21
+ from baseaicore import UNSUPPORTED, is_supported
22
+ from pydantic import Field, model_validator
23
+
24
+ from setspec.base import PayloadDefinition, WireEnum, payload_models
25
+ from setspec.serialization import MeasurementField
26
+
27
+ __all__ = [
28
+ "Aggregation",
29
+ "MetricValueFields",
30
+ "MetricValueIn",
31
+ "MetricValueOut",
32
+ ]
33
+
34
+ _MINIMUM_SAMPLES_FOR_DISPERSION = 2
35
+
36
+ _METRIC_KEY_PATTERN = r"^[a-z][a-z0-9_]*(\.[a-z][a-z0-9_]*)*$"
37
+ """Lower snake case, starting with a letter, optionally dot-separated into segments.
38
+
39
+ Enforced rather than documented because a metric key is an identifier a consumer *matches on*:
40
+ ``ttft_ms`` and ``TTFT_ms`` from two producers would be two metrics to every reader and one metric
41
+ to every author. The suite's own rule — same concept, same name, everywhere — needs the wire format
42
+ to be the place it cannot be broken.
43
+
44
+ **Dots are for namespacing, and a real producer needs them**: FreeWeight's user-authored goal suites
45
+ emit one metric per criterion as ``criterion.<key>``, where the segment after the dot is the
46
+ author's own slug. A flat pattern would either reject those keys or push the producer to flatten
47
+ them into ``criterion_<key>``, which collides with a criterion actually named ``key``. Each segment
48
+ follows the same rule as a whole key, so ``criterion.house_voice`` is legal and
49
+ ``criterion.House Voice`` is not."""
50
+
51
+
52
+ class Aggregation(StrEnum):
53
+ """How the samples behind a metric were reduced to one number.
54
+
55
+ A ``StrEnum`` so it serializes and logs as its own name rather than an opaque integer
56
+ (coding standards §2). The value is what appears on the wire.
57
+
58
+ Recording this is not bookkeeping: a mean and a p95 of the same samples answer different
59
+ questions, and a consumer that compares one against the other is comparing nothing. The
60
+ producer knows which it computed; the wire format makes it say so.
61
+ """
62
+
63
+ SINGLE = "single"
64
+ """One observation, not aggregated. ``sample_count`` is 1 and dispersion is unsupported."""
65
+
66
+ MEAN = "mean"
67
+ """Arithmetic mean of the supported samples."""
68
+
69
+ MEDIAN = "median"
70
+ """50th percentile — the value to prefer over ``MEAN`` where outliers are expected."""
71
+
72
+ MIN = "min"
73
+ """Smallest supported sample."""
74
+
75
+ MAX = "max"
76
+ """Largest supported sample."""
77
+
78
+ SUM = "sum"
79
+ """Total across the supported samples."""
80
+
81
+ COUNT = "count"
82
+ """How many events occurred — a tally, distinct from ``sample_count``, which says how many
83
+ observations produced this statistic."""
84
+
85
+ STDDEV = "stddev"
86
+ """Standard deviation of the supported samples, reported as the value in its own right."""
87
+
88
+ P50 = "p50"
89
+ """50th percentile — the same statistic as ``MEDIAN``, spelled the way a percentile family
90
+ (``p50``/``p95``/``p99``) is usually reported together; both members exist so a producer never
91
+ has to translate its own vocabulary to satisfy this one."""
92
+
93
+ P95 = "p95"
94
+ """95th percentile."""
95
+
96
+ P99 = "p99"
97
+ """99th percentile."""
98
+
99
+ RATIO = "ratio"
100
+ """A proportion in ``[0, 1]`` that is not a mean of pass/fail samples but a direct ratio —
101
+ e.g. a memory-overhead ratio computed from two other measurements."""
102
+
103
+ RAW = "raw"
104
+ """A single unaggregated reading, kept distinct from ``SINGLE``: ``SINGLE`` still promises a
105
+ real, comparable measurement of the metric's stated unit, while ``RAW`` marks a value passed
106
+ through without this build knowing whether it was ever meant to be aggregated at all."""
107
+
108
+
109
+ class MetricValueFields(PayloadDefinition):
110
+ """Field definitions for ``metric.value``; use :data:`MetricValueOut` / :data:`MetricValueIn`.
111
+
112
+ A measured quantity, the statistic that produced it, and enough context for a consumer to know
113
+ whether comparing it to another one is meaningful.
114
+
115
+ Attributes:
116
+ metric_key: Which metric this is — ``"decode_tokens_per_second"``, ``"task_success"``.
117
+ Required, because both :class:`~setspec.benchmark.v1.BenchmarkResultFields` and
118
+ :class:`~setspec.benchmark.v1.BenchmarkRunSummaryFields` carry *sequences* of this
119
+ model: without a key, a consumer receives a list of numbers it cannot attribute,
120
+ chart, compare across runs, or check for the metric it was looking for. Lower snake
121
+ case, so one metric has one spelling everywhere in the suite.
122
+ value: The measurement, or ``UNSUPPORTED`` when this environment could not provide one.
123
+ Never ``null`` and never ``0`` as a stand-in for absence
124
+ (ADR-0016 §4).
125
+ unit: The unit the value is in — ``"ms"``, ``"tokens_per_second"``, ``"bytes"``,
126
+ ``"ratio"``. Required and non-empty: a number whose unit lives only in a field name
127
+ somewhere upstream is a number that will eventually be compared against a different
128
+ one (coding standards §3). Dimensionless quantities say so explicitly rather than
129
+ passing an empty string.
130
+ aggregation: Which statistic ``value`` is.
131
+ higher_is_better: Whether a larger value is a better result. Carried per metric because
132
+ the answer differs between metrics in the same payload — throughput and latency point
133
+ in opposite directions — and a consumer ranking results cannot infer it from the unit.
134
+ sample_count: How many **supported** samples produced ``value``. Unsupported samples are
135
+ excluded from the statistic and from this count (ADR-0016 §6), so it is the honest
136
+ denominator, not the number of attempts.
137
+ dispersion: Spread of those samples, as a standard deviation in the same unit as
138
+ ``value``. ``UNSUPPORTED`` when fewer than two supported samples exist, because the
139
+ spread of a single observation is undefined rather than zero.
140
+ """
141
+
142
+ metric_key: str = Field(min_length=1, pattern=_METRIC_KEY_PATTERN)
143
+ value: MeasurementField
144
+ unit: str = Field(min_length=1)
145
+ aggregation: WireEnum[Aggregation]
146
+ higher_is_better: bool
147
+ sample_count: int = Field(ge=0)
148
+ dispersion: MeasurementField
149
+
150
+ @model_validator(mode="after")
151
+ def _check_sample_coherence(self) -> Self:
152
+ """Enforce ADR-0016 §6: the value and its sample count must tell the same story.
153
+
154
+ Raises:
155
+ ValueError: If a real value claims no samples, if an unsupported value claims some, or
156
+ if a dispersion is reported for fewer than two samples.
157
+ """
158
+ if is_supported(self.value) and self.sample_count < 1:
159
+ raise ValueError(
160
+ "a metric with a real value must report at least one supported sample; "
161
+ "sample_count=0 with a number in `value` means the number came from nowhere "
162
+ "(ADR-0016 §6)"
163
+ )
164
+ if not is_supported(self.value) and self.sample_count != 0:
165
+ raise ValueError(
166
+ f"an unsupported metric has no supported samples, but sample_count is "
167
+ f"{self.sample_count}. A metric with no supported samples is itself unsupported — "
168
+ "report the attempts elsewhere, not as the denominator of a statistic that was "
169
+ "never computed (ADR-0016 §6)"
170
+ )
171
+ if (
172
+ self.dispersion is not UNSUPPORTED
173
+ and self.sample_count < _MINIMUM_SAMPLES_FOR_DISPERSION
174
+ ):
175
+ raise ValueError(
176
+ f"dispersion needs at least {_MINIMUM_SAMPLES_FOR_DISPERSION} supported samples; "
177
+ f"sample_count is {self.sample_count}. The spread of a single observation is "
178
+ "undefined, not zero — report it as 'unsupported'"
179
+ )
180
+ return self
181
+
182
+
183
+ MetricValueOut, MetricValueIn = payload_models(MetricValueFields)
184
+ """The ``metric.value`` payload pair: ``Out`` for writers, ``In`` for readers."""
setspec/model/v1.py ADDED
@@ -0,0 +1,169 @@
1
+ """Contract module — ``model.identity`` v1: which weights, plus what a provider says about them.
2
+
3
+ Imports pydantic and :mod:`baseaicore`; performs no I/O. Exchange form of
4
+ :class:`baseaicore.ModelIdentity` (the identity triple) and :class:`baseaicore.ModelDescriptor`
5
+ (the refreshable metadata a provider reports about those weights), combined into one payload
6
+ because ADR-0022 §1 always
7
+ carries them together as ``capability.evidence.model``.
8
+
9
+ **Status: draft (`1.0`).** Registered in :data:`setspec.envelope.SUPPORTED_SCHEMAS` so FreeWeight
10
+ has a concrete model to build against, but not yet frozen — see
11
+ [development plan Phase 2](../../../docs/packages/setspec/development-plan.md) and
12
+ [Phase 4](../../../docs/packages/setspec/development-plan.md), which promotes this to `1.0` only
13
+ after FreeWeight has produced real results against it. A field may still be added, tightened or
14
+ reshaped by that promotion without a major version bump signalling it in advance.
15
+
16
+ Deliberately omitted: :attr:`baseaicore.ModelDescriptor.raw`, the untouched provider response.
17
+ It carries no contract — its own docstring says nothing above the normalizer may read it for
18
+ business logic — and freezing its presence on the wire would promise a shape for a value this
19
+ package cannot describe.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ from typing import Any, Self
25
+
26
+ from baseaicore import (
27
+ UNSUPPORTED,
28
+ IdentityConfidence,
29
+ ModelCapabilityFlag,
30
+ ModelIdentity,
31
+ ProviderKind,
32
+ )
33
+ from baseaicore import ValidationError as SuiteValidationError
34
+ from pydantic import Field, model_validator
35
+
36
+ from setspec.base import PayloadDefinition, WireEnum, WireSequence, payload_models
37
+ from setspec.serialization import MeasurementField, TimestampField
38
+
39
+ __all__ = [
40
+ "ModelIdentityFields",
41
+ "ModelIdentityIn",
42
+ "ModelIdentityOut",
43
+ ]
44
+
45
+
46
+ class ModelIdentityFields(PayloadDefinition):
47
+ """Field definitions for ``model.identity``; use :data:`ModelIdentityOut` /
48
+ :data:`ModelIdentityIn`.
49
+
50
+ :attr:`canonical_id` and :attr:`identity_confidence` are materialized rather than left for a
51
+ reader to derive, so a consumer can display or index by them without reconstructing a
52
+ :class:`baseaicore.ModelIdentity`. Both are still validated against the identity triple on
53
+ every construction — see :meth:`_check_identity_coherence` — so "provenance completeness
54
+ enforced by the schema" (Phase 2 gold standard) applies to the derived fields too, not only
55
+ the fields nothing computes.
56
+
57
+ Attributes:
58
+ provider_kind: Which kind of provider serves these weights ([ADR-0008]
59
+ (0008 canonical model identity)). Reuses
60
+ :class:`baseaicore.ProviderKind` directly rather than a shadow enum, so a new provider
61
+ kind reaches this schema the moment BaseAiCore adds it.
62
+ provider_model_name: Exactly as the provider names it, case and punctuation preserved.
63
+ artifact_digest: ``"sha256:"`` + 64 lowercase hex characters, or ``None`` when the
64
+ provider exposes none.
65
+ identity_confidence: ``digest`` iff ``artifact_digest`` is present, ``name_only``
66
+ otherwise — checked, not merely documented.
67
+ canonical_id: ``{provider_kind}/{provider_model_name}@{digest_short}``
68
+ (ADR-0024). Lossy and
69
+ display-only; never parsed back into its parts.
70
+ observed_at: When this descriptor snapshot was read from the provider.
71
+ family: The model family name, e.g. ``"qwen3.5"``.
72
+ architecture: The architecture name, e.g. ``"transformer"``, ``"mamba"``.
73
+ parameter_count: Total parameter count.
74
+ active_parameter_count: MoE active parameters per token; equal to
75
+ ``parameter_count`` for a dense model.
76
+ expert_count: Number of experts, for a mixture-of-experts model.
77
+ quantization: Weight quantization, e.g. ``"Q8_0"``.
78
+ weight_format: File format, e.g. ``"gguf"``, ``"safetensors"``.
79
+ size_bytes: On-disk size of the weights.
80
+ max_context: The context length the model *advertises* — not the context a provider is
81
+ configured to serve, which is ``execution.served_context`` on a benchmark result
82
+ (ADR-0023 §4).
83
+ embedding_dim: Hidden/embedding dimension.
84
+ layers: Transformer layer count.
85
+ attention_heads: Attention head count.
86
+ kv_heads: Key/value head count.
87
+ head_dim: Dimension of each attention head.
88
+ vocab_size: Tokenizer vocabulary size.
89
+ rope_config: RoPE scaling configuration, in the provider's own shape.
90
+ sliding_window: Sliding-attention window size, if the architecture uses one.
91
+ declared_capabilities: What the provider *claims* this model can do — never conflated
92
+ with a measured capability from ``capability.evidence``.
93
+ license_text: The model's license, if the provider exposes one.
94
+ """
95
+
96
+ provider_kind: WireEnum[ProviderKind]
97
+ provider_model_name: str = Field(min_length=1)
98
+ artifact_digest: str | None = None
99
+ identity_confidence: WireEnum[IdentityConfidence]
100
+ canonical_id: str = Field(min_length=1)
101
+
102
+ observed_at: TimestampField
103
+ family: str | None = None
104
+ architecture: str | None = None
105
+ parameter_count: MeasurementField = UNSUPPORTED
106
+ active_parameter_count: MeasurementField = UNSUPPORTED
107
+ expert_count: MeasurementField = UNSUPPORTED
108
+ quantization: str | None = None
109
+ weight_format: str | None = None
110
+ size_bytes: MeasurementField = UNSUPPORTED
111
+ max_context: MeasurementField = UNSUPPORTED
112
+ embedding_dim: MeasurementField = UNSUPPORTED
113
+ layers: MeasurementField = UNSUPPORTED
114
+ attention_heads: MeasurementField = UNSUPPORTED
115
+ kv_heads: MeasurementField = UNSUPPORTED
116
+ head_dim: MeasurementField = UNSUPPORTED
117
+ vocab_size: MeasurementField = UNSUPPORTED
118
+ rope_config: dict[str, Any] | None = None
119
+ sliding_window: MeasurementField = UNSUPPORTED
120
+ declared_capabilities: WireSequence[WireEnum[ModelCapabilityFlag]] = ()
121
+ license_text: str | None = None
122
+
123
+ @model_validator(mode="after")
124
+ def _check_identity_coherence(self) -> Self:
125
+ """Recompute the identity triple's derived fields and require them to agree.
126
+
127
+ Unlike a machine fingerprint — whose inclusion policy is allowed to change under a
128
+ profile that must still reconstruct exactly as it was written — a canonical ID and an
129
+ identity confidence are pure functions of the triple with no such historical caveat
130
+ (:class:`baseaicore.ModelIdentity` never re-verifies a stored fingerprint for that reason,
131
+ but always recomputes ``canonical_id``). Recomputing here is therefore safe, not merely
132
+ convenient, and it catches a producer that materialized the derived fields inconsistently
133
+ with the triple that stands next to them.
134
+
135
+ Raises:
136
+ ValueError: If ``provider_model_name`` or ``artifact_digest`` fails
137
+ :class:`baseaicore.ModelIdentity`'s own validation, or if ``canonical_id`` or
138
+ ``identity_confidence`` disagrees with what the triple recomputes.
139
+ """
140
+ try:
141
+ identity = ModelIdentity(
142
+ provider_kind=self.provider_kind,
143
+ provider_model_name=self.provider_model_name,
144
+ artifact_digest=self.artifact_digest,
145
+ )
146
+ except SuiteValidationError as exc:
147
+ # baseaicore.ValidationError is a SuiteError, not a ValueError; pydantic only
148
+ # aggregates ValueError/AssertionError into its own report (setspec.serialization
149
+ # hits the same seam for timestamps).
150
+ raise ValueError(str(exc)) from exc
151
+ if identity.canonical_id != self.canonical_id:
152
+ raise ValueError(
153
+ f"canonical_id {self.canonical_id!r} does not match the identity triple, which "
154
+ f"recomputes to {identity.canonical_id!r}. canonical_id is a pure function of "
155
+ "provider_kind, provider_model_name and artifact_digest (ADR-0024) — it is "
156
+ "carried on the wire for convenience, not as an independent fact."
157
+ )
158
+ if identity.identity_confidence.value != self.identity_confidence:
159
+ raise ValueError(
160
+ f"identity_confidence {self.identity_confidence!r} does not match the identity "
161
+ f"triple: artifact_digest is "
162
+ f"{'present' if self.artifact_digest is not None else 'absent'}, which makes "
163
+ f"this identity {identity.identity_confidence.value!r}."
164
+ )
165
+ return self
166
+
167
+
168
+ ModelIdentityOut, ModelIdentityIn = payload_models(ModelIdentityFields)
169
+ """The ``model.identity`` payload pair: ``Out`` for writers, ``In`` for readers."""
setspec/provenance.py ADDED
@@ -0,0 +1,59 @@
1
+ """Contract module — provenance building blocks shared by more than one versioned payload.
2
+
3
+ Imports pydantic and :mod:`baseaicore`; performs no I/O.
4
+
5
+ This module exists for the same reason :mod:`setspec.metrics` does. A sub-model used by two
6
+ payload types belongs to neither of them: if ``EnvironmentFields`` lived in
7
+ :mod:`setspec.benchmark.v1`, then the day ``benchmark.result`` needs a changed environment shape,
8
+ whoever edits that class would silently change ``capability.evidence`` v1 too — a frozen contract
9
+ mutating because someone edited a different payload's module, which is precisely the drift
10
+ ADR-0009 exists to prevent. A shared block gets
11
+ a neutral home so that changing it is an obviously cross-cutting act rather than an accident.
12
+
13
+ Nothing here is a wire payload in its own right: none of these names appears in
14
+ :data:`~setspec.envelope.SUPPORTED_SCHEMAS`, and none is generated into an ``Out``/``In`` pair.
15
+ They are field groups that versioned payloads embed, which is why they carry no version of their
16
+ own — their version is whichever payload's version they appear inside.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ from baseaicore import ProviderKind
22
+ from pydantic import Field
23
+
24
+ from setspec.base import PayloadDefinition, WireEnum
25
+
26
+ __all__ = ["EnvironmentFields"]
27
+
28
+
29
+ class EnvironmentFields(PayloadDefinition):
30
+ """Provider and drift-sensitive environment facts at the moment of measurement.
31
+
32
+ Unifies two bullets that
33
+ Machine Identity §6 and
34
+ ADR-0022 §1 describe separately
35
+ but with identical content — "provider kind + provider version" and "GPU driver, CUDA, OS
36
+ version at measurement" — into the one nested object ADR-0022 already names ``environment`` on
37
+ ``capability.evidence``, so a benchmark result and the evidence aggregated from it carry
38
+ environment facts in the same shape rather than two shapes a consumer has to reconcile.
39
+
40
+ Every field but the provider's own identity is a **drift signal, not identity**: a driver
41
+ upgrade or an OS patch must never re-identify a machine (that is the machine fingerprint's
42
+ job, and it deliberately excludes all of these), but it does reduce confidence in performance
43
+ evidence measured before it
44
+ (ADR-0017's
45
+ ``environment_factor``). Recording them is what makes that reduction computable at all.
46
+
47
+ Attributes:
48
+ provider_kind: Which kind of provider served the model for this measurement.
49
+ provider_version: The provider's own version string, e.g. ``"0.32.13"``.
50
+ gpu_driver_version: A drift signal; ``None`` when no GPU was involved.
51
+ cuda_version: The CUDA/ROCm toolkit version. A drift signal.
52
+ os_version: A drift signal.
53
+ """
54
+
55
+ provider_kind: WireEnum[ProviderKind]
56
+ provider_version: str = Field(min_length=1)
57
+ gpu_driver_version: str | None = None
58
+ cuda_version: str | None = None
59
+ os_version: str | None = None
setspec/py.typed ADDED
File without changes
File without changes