setspec 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- setspec/__about__.py +3 -0
- setspec/__init__.py +114 -0
- setspec/artifacts.py +4 -0
- setspec/base.py +256 -0
- setspec/benchmark/v1.py +594 -0
- setspec/capability/v1.py +394 -0
- setspec/envelope.py +468 -0
- setspec/error/v1.py +4 -0
- setspec/errors.py +68 -0
- setspec/event/v1.py +4 -0
- setspec/goal/v1.py +390 -0
- setspec/goldens/.gitkeep +0 -0
- setspec/machine/v1.py +131 -0
- setspec/metrics.py +184 -0
- setspec/model/v1.py +169 -0
- setspec/provenance.py +59 -0
- setspec/py.typed +0 -0
- setspec/schemas/.gitkeep +0 -0
- setspec/serialization.py +330 -0
- setspec/vocabulary.py +205 -0
- setspec-0.2.0.dist-info/METADATA +160 -0
- setspec-0.2.0.dist-info/RECORD +24 -0
- setspec-0.2.0.dist-info/WHEEL +4 -0
- setspec-0.2.0.dist-info/licenses/LICENSE +201 -0
setspec/metrics.py
ADDED
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
"""Contract module — the measured value that every benchmark metric reduces to.
|
|
2
|
+
|
|
3
|
+
Imports pydantic and :mod:`baseaicore`; performs no I/O and computes nothing. SetSpec carries
|
|
4
|
+
statistics, it never produces them: the aggregation named here was performed by the producer, and
|
|
5
|
+
this model's job is to make the producer state what it did rather than hand over a bare number
|
|
6
|
+
(spec §3, non-goals).
|
|
7
|
+
|
|
8
|
+
The invariants below are the schema-level half of
|
|
9
|
+
ADR-0016 §6. That ADR requires unsupported
|
|
10
|
+
samples to be *excluded* from a statistic and the surviving sample count to be reported next to it;
|
|
11
|
+
a model that let a caller write ``value=0.0, sample_count=0`` would make the rule advisory. Here it
|
|
12
|
+
is structural — the two fields cannot disagree, so a metric that was never measured cannot be
|
|
13
|
+
serialized as one that measured zero.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from enum import StrEnum
|
|
19
|
+
from typing import Self
|
|
20
|
+
|
|
21
|
+
from baseaicore import UNSUPPORTED, is_supported
|
|
22
|
+
from pydantic import Field, model_validator
|
|
23
|
+
|
|
24
|
+
from setspec.base import PayloadDefinition, WireEnum, payload_models
|
|
25
|
+
from setspec.serialization import MeasurementField
|
|
26
|
+
|
|
27
|
+
__all__ = [
|
|
28
|
+
"Aggregation",
|
|
29
|
+
"MetricValueFields",
|
|
30
|
+
"MetricValueIn",
|
|
31
|
+
"MetricValueOut",
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
_MINIMUM_SAMPLES_FOR_DISPERSION = 2
|
|
35
|
+
|
|
36
|
+
_METRIC_KEY_PATTERN = r"^[a-z][a-z0-9_]*(\.[a-z][a-z0-9_]*)*$"
|
|
37
|
+
"""Lower snake case, starting with a letter, optionally dot-separated into segments.
|
|
38
|
+
|
|
39
|
+
Enforced rather than documented because a metric key is an identifier a consumer *matches on*:
|
|
40
|
+
``ttft_ms`` and ``TTFT_ms`` from two producers would be two metrics to every reader and one metric
|
|
41
|
+
to every author. The suite's own rule — same concept, same name, everywhere — needs the wire format
|
|
42
|
+
to be the place it cannot be broken.
|
|
43
|
+
|
|
44
|
+
**Dots are for namespacing, and a real producer needs them**: FreeWeight's user-authored goal suites
|
|
45
|
+
emit one metric per criterion as ``criterion.<key>``, where the segment after the dot is the
|
|
46
|
+
author's own slug. A flat pattern would either reject those keys or push the producer to flatten
|
|
47
|
+
them into ``criterion_<key>``, which collides with a criterion actually named ``key``. Each segment
|
|
48
|
+
follows the same rule as a whole key, so ``criterion.house_voice`` is legal and
|
|
49
|
+
``criterion.House Voice`` is not."""
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class Aggregation(StrEnum):
|
|
53
|
+
"""How the samples behind a metric were reduced to one number.
|
|
54
|
+
|
|
55
|
+
A ``StrEnum`` so it serializes and logs as its own name rather than an opaque integer
|
|
56
|
+
(coding standards §2). The value is what appears on the wire.
|
|
57
|
+
|
|
58
|
+
Recording this is not bookkeeping: a mean and a p95 of the same samples answer different
|
|
59
|
+
questions, and a consumer that compares one against the other is comparing nothing. The
|
|
60
|
+
producer knows which it computed; the wire format makes it say so.
|
|
61
|
+
"""
|
|
62
|
+
|
|
63
|
+
SINGLE = "single"
|
|
64
|
+
"""One observation, not aggregated. ``sample_count`` is 1 and dispersion is unsupported."""
|
|
65
|
+
|
|
66
|
+
MEAN = "mean"
|
|
67
|
+
"""Arithmetic mean of the supported samples."""
|
|
68
|
+
|
|
69
|
+
MEDIAN = "median"
|
|
70
|
+
"""50th percentile — the value to prefer over ``MEAN`` where outliers are expected."""
|
|
71
|
+
|
|
72
|
+
MIN = "min"
|
|
73
|
+
"""Smallest supported sample."""
|
|
74
|
+
|
|
75
|
+
MAX = "max"
|
|
76
|
+
"""Largest supported sample."""
|
|
77
|
+
|
|
78
|
+
SUM = "sum"
|
|
79
|
+
"""Total across the supported samples."""
|
|
80
|
+
|
|
81
|
+
COUNT = "count"
|
|
82
|
+
"""How many events occurred — a tally, distinct from ``sample_count``, which says how many
|
|
83
|
+
observations produced this statistic."""
|
|
84
|
+
|
|
85
|
+
STDDEV = "stddev"
|
|
86
|
+
"""Standard deviation of the supported samples, reported as the value in its own right."""
|
|
87
|
+
|
|
88
|
+
P50 = "p50"
|
|
89
|
+
"""50th percentile — the same statistic as ``MEDIAN``, spelled the way a percentile family
|
|
90
|
+
(``p50``/``p95``/``p99``) is usually reported together; both members exist so a producer never
|
|
91
|
+
has to translate its own vocabulary to satisfy this one."""
|
|
92
|
+
|
|
93
|
+
P95 = "p95"
|
|
94
|
+
"""95th percentile."""
|
|
95
|
+
|
|
96
|
+
P99 = "p99"
|
|
97
|
+
"""99th percentile."""
|
|
98
|
+
|
|
99
|
+
RATIO = "ratio"
|
|
100
|
+
"""A proportion in ``[0, 1]`` that is not a mean of pass/fail samples but a direct ratio —
|
|
101
|
+
e.g. a memory-overhead ratio computed from two other measurements."""
|
|
102
|
+
|
|
103
|
+
RAW = "raw"
|
|
104
|
+
"""A single unaggregated reading, kept distinct from ``SINGLE``: ``SINGLE`` still promises a
|
|
105
|
+
real, comparable measurement of the metric's stated unit, while ``RAW`` marks a value passed
|
|
106
|
+
through without this build knowing whether it was ever meant to be aggregated at all."""
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
class MetricValueFields(PayloadDefinition):
|
|
110
|
+
"""Field definitions for ``metric.value``; use :data:`MetricValueOut` / :data:`MetricValueIn`.
|
|
111
|
+
|
|
112
|
+
A measured quantity, the statistic that produced it, and enough context for a consumer to know
|
|
113
|
+
whether comparing it to another one is meaningful.
|
|
114
|
+
|
|
115
|
+
Attributes:
|
|
116
|
+
metric_key: Which metric this is — ``"decode_tokens_per_second"``, ``"task_success"``.
|
|
117
|
+
Required, because both :class:`~setspec.benchmark.v1.BenchmarkResultFields` and
|
|
118
|
+
:class:`~setspec.benchmark.v1.BenchmarkRunSummaryFields` carry *sequences* of this
|
|
119
|
+
model: without a key, a consumer receives a list of numbers it cannot attribute,
|
|
120
|
+
chart, compare across runs, or check for the metric it was looking for. Lower snake
|
|
121
|
+
case, so one metric has one spelling everywhere in the suite.
|
|
122
|
+
value: The measurement, or ``UNSUPPORTED`` when this environment could not provide one.
|
|
123
|
+
Never ``null`` and never ``0`` as a stand-in for absence
|
|
124
|
+
(ADR-0016 §4).
|
|
125
|
+
unit: The unit the value is in — ``"ms"``, ``"tokens_per_second"``, ``"bytes"``,
|
|
126
|
+
``"ratio"``. Required and non-empty: a number whose unit lives only in a field name
|
|
127
|
+
somewhere upstream is a number that will eventually be compared against a different
|
|
128
|
+
one (coding standards §3). Dimensionless quantities say so explicitly rather than
|
|
129
|
+
passing an empty string.
|
|
130
|
+
aggregation: Which statistic ``value`` is.
|
|
131
|
+
higher_is_better: Whether a larger value is a better result. Carried per metric because
|
|
132
|
+
the answer differs between metrics in the same payload — throughput and latency point
|
|
133
|
+
in opposite directions — and a consumer ranking results cannot infer it from the unit.
|
|
134
|
+
sample_count: How many **supported** samples produced ``value``. Unsupported samples are
|
|
135
|
+
excluded from the statistic and from this count (ADR-0016 §6), so it is the honest
|
|
136
|
+
denominator, not the number of attempts.
|
|
137
|
+
dispersion: Spread of those samples, as a standard deviation in the same unit as
|
|
138
|
+
``value``. ``UNSUPPORTED`` when fewer than two supported samples exist, because the
|
|
139
|
+
spread of a single observation is undefined rather than zero.
|
|
140
|
+
"""
|
|
141
|
+
|
|
142
|
+
metric_key: str = Field(min_length=1, pattern=_METRIC_KEY_PATTERN)
|
|
143
|
+
value: MeasurementField
|
|
144
|
+
unit: str = Field(min_length=1)
|
|
145
|
+
aggregation: WireEnum[Aggregation]
|
|
146
|
+
higher_is_better: bool
|
|
147
|
+
sample_count: int = Field(ge=0)
|
|
148
|
+
dispersion: MeasurementField
|
|
149
|
+
|
|
150
|
+
@model_validator(mode="after")
|
|
151
|
+
def _check_sample_coherence(self) -> Self:
|
|
152
|
+
"""Enforce ADR-0016 §6: the value and its sample count must tell the same story.
|
|
153
|
+
|
|
154
|
+
Raises:
|
|
155
|
+
ValueError: If a real value claims no samples, if an unsupported value claims some, or
|
|
156
|
+
if a dispersion is reported for fewer than two samples.
|
|
157
|
+
"""
|
|
158
|
+
if is_supported(self.value) and self.sample_count < 1:
|
|
159
|
+
raise ValueError(
|
|
160
|
+
"a metric with a real value must report at least one supported sample; "
|
|
161
|
+
"sample_count=0 with a number in `value` means the number came from nowhere "
|
|
162
|
+
"(ADR-0016 §6)"
|
|
163
|
+
)
|
|
164
|
+
if not is_supported(self.value) and self.sample_count != 0:
|
|
165
|
+
raise ValueError(
|
|
166
|
+
f"an unsupported metric has no supported samples, but sample_count is "
|
|
167
|
+
f"{self.sample_count}. A metric with no supported samples is itself unsupported — "
|
|
168
|
+
"report the attempts elsewhere, not as the denominator of a statistic that was "
|
|
169
|
+
"never computed (ADR-0016 §6)"
|
|
170
|
+
)
|
|
171
|
+
if (
|
|
172
|
+
self.dispersion is not UNSUPPORTED
|
|
173
|
+
and self.sample_count < _MINIMUM_SAMPLES_FOR_DISPERSION
|
|
174
|
+
):
|
|
175
|
+
raise ValueError(
|
|
176
|
+
f"dispersion needs at least {_MINIMUM_SAMPLES_FOR_DISPERSION} supported samples; "
|
|
177
|
+
f"sample_count is {self.sample_count}. The spread of a single observation is "
|
|
178
|
+
"undefined, not zero — report it as 'unsupported'"
|
|
179
|
+
)
|
|
180
|
+
return self
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
MetricValueOut, MetricValueIn = payload_models(MetricValueFields)
|
|
184
|
+
"""The ``metric.value`` payload pair: ``Out`` for writers, ``In`` for readers."""
|
setspec/model/v1.py
ADDED
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
"""Contract module — ``model.identity`` v1: which weights, plus what a provider says about them.
|
|
2
|
+
|
|
3
|
+
Imports pydantic and :mod:`baseaicore`; performs no I/O. Exchange form of
|
|
4
|
+
:class:`baseaicore.ModelIdentity` (the identity triple) and :class:`baseaicore.ModelDescriptor`
|
|
5
|
+
(the refreshable metadata a provider reports about those weights), combined into one payload
|
|
6
|
+
because ADR-0022 §1 always
|
|
7
|
+
carries them together as ``capability.evidence.model``.
|
|
8
|
+
|
|
9
|
+
**Status: draft (`1.0`).** Registered in :data:`setspec.envelope.SUPPORTED_SCHEMAS` so FreeWeight
|
|
10
|
+
has a concrete model to build against, but not yet frozen — see
|
|
11
|
+
[development plan Phase 2](../../../docs/packages/setspec/development-plan.md) and
|
|
12
|
+
[Phase 4](../../../docs/packages/setspec/development-plan.md), which promotes this to `1.0` only
|
|
13
|
+
after FreeWeight has produced real results against it. A field may still be added, tightened or
|
|
14
|
+
reshaped by that promotion without a major version bump signalling it in advance.
|
|
15
|
+
|
|
16
|
+
Deliberately omitted: :attr:`baseaicore.ModelDescriptor.raw`, the untouched provider response.
|
|
17
|
+
It carries no contract — its own docstring says nothing above the normalizer may read it for
|
|
18
|
+
business logic — and freezing its presence on the wire would promise a shape for a value this
|
|
19
|
+
package cannot describe.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
from typing import Any, Self
|
|
25
|
+
|
|
26
|
+
from baseaicore import (
|
|
27
|
+
UNSUPPORTED,
|
|
28
|
+
IdentityConfidence,
|
|
29
|
+
ModelCapabilityFlag,
|
|
30
|
+
ModelIdentity,
|
|
31
|
+
ProviderKind,
|
|
32
|
+
)
|
|
33
|
+
from baseaicore import ValidationError as SuiteValidationError
|
|
34
|
+
from pydantic import Field, model_validator
|
|
35
|
+
|
|
36
|
+
from setspec.base import PayloadDefinition, WireEnum, WireSequence, payload_models
|
|
37
|
+
from setspec.serialization import MeasurementField, TimestampField
|
|
38
|
+
|
|
39
|
+
__all__ = [
|
|
40
|
+
"ModelIdentityFields",
|
|
41
|
+
"ModelIdentityIn",
|
|
42
|
+
"ModelIdentityOut",
|
|
43
|
+
]
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class ModelIdentityFields(PayloadDefinition):
|
|
47
|
+
"""Field definitions for ``model.identity``; use :data:`ModelIdentityOut` /
|
|
48
|
+
:data:`ModelIdentityIn`.
|
|
49
|
+
|
|
50
|
+
:attr:`canonical_id` and :attr:`identity_confidence` are materialized rather than left for a
|
|
51
|
+
reader to derive, so a consumer can display or index by them without reconstructing a
|
|
52
|
+
:class:`baseaicore.ModelIdentity`. Both are still validated against the identity triple on
|
|
53
|
+
every construction — see :meth:`_check_identity_coherence` — so "provenance completeness
|
|
54
|
+
enforced by the schema" (Phase 2 gold standard) applies to the derived fields too, not only
|
|
55
|
+
the fields nothing computes.
|
|
56
|
+
|
|
57
|
+
Attributes:
|
|
58
|
+
provider_kind: Which kind of provider serves these weights ([ADR-0008]
|
|
59
|
+
(0008 canonical model identity)). Reuses
|
|
60
|
+
:class:`baseaicore.ProviderKind` directly rather than a shadow enum, so a new provider
|
|
61
|
+
kind reaches this schema the moment BaseAiCore adds it.
|
|
62
|
+
provider_model_name: Exactly as the provider names it, case and punctuation preserved.
|
|
63
|
+
artifact_digest: ``"sha256:"`` + 64 lowercase hex characters, or ``None`` when the
|
|
64
|
+
provider exposes none.
|
|
65
|
+
identity_confidence: ``digest`` iff ``artifact_digest`` is present, ``name_only``
|
|
66
|
+
otherwise — checked, not merely documented.
|
|
67
|
+
canonical_id: ``{provider_kind}/{provider_model_name}@{digest_short}``
|
|
68
|
+
(ADR-0024). Lossy and
|
|
69
|
+
display-only; never parsed back into its parts.
|
|
70
|
+
observed_at: When this descriptor snapshot was read from the provider.
|
|
71
|
+
family: The model family name, e.g. ``"qwen3.5"``.
|
|
72
|
+
architecture: The architecture name, e.g. ``"transformer"``, ``"mamba"``.
|
|
73
|
+
parameter_count: Total parameter count.
|
|
74
|
+
active_parameter_count: MoE active parameters per token; equal to
|
|
75
|
+
``parameter_count`` for a dense model.
|
|
76
|
+
expert_count: Number of experts, for a mixture-of-experts model.
|
|
77
|
+
quantization: Weight quantization, e.g. ``"Q8_0"``.
|
|
78
|
+
weight_format: File format, e.g. ``"gguf"``, ``"safetensors"``.
|
|
79
|
+
size_bytes: On-disk size of the weights.
|
|
80
|
+
max_context: The context length the model *advertises* — not the context a provider is
|
|
81
|
+
configured to serve, which is ``execution.served_context`` on a benchmark result
|
|
82
|
+
(ADR-0023 §4).
|
|
83
|
+
embedding_dim: Hidden/embedding dimension.
|
|
84
|
+
layers: Transformer layer count.
|
|
85
|
+
attention_heads: Attention head count.
|
|
86
|
+
kv_heads: Key/value head count.
|
|
87
|
+
head_dim: Dimension of each attention head.
|
|
88
|
+
vocab_size: Tokenizer vocabulary size.
|
|
89
|
+
rope_config: RoPE scaling configuration, in the provider's own shape.
|
|
90
|
+
sliding_window: Sliding-attention window size, if the architecture uses one.
|
|
91
|
+
declared_capabilities: What the provider *claims* this model can do — never conflated
|
|
92
|
+
with a measured capability from ``capability.evidence``.
|
|
93
|
+
license_text: The model's license, if the provider exposes one.
|
|
94
|
+
"""
|
|
95
|
+
|
|
96
|
+
provider_kind: WireEnum[ProviderKind]
|
|
97
|
+
provider_model_name: str = Field(min_length=1)
|
|
98
|
+
artifact_digest: str | None = None
|
|
99
|
+
identity_confidence: WireEnum[IdentityConfidence]
|
|
100
|
+
canonical_id: str = Field(min_length=1)
|
|
101
|
+
|
|
102
|
+
observed_at: TimestampField
|
|
103
|
+
family: str | None = None
|
|
104
|
+
architecture: str | None = None
|
|
105
|
+
parameter_count: MeasurementField = UNSUPPORTED
|
|
106
|
+
active_parameter_count: MeasurementField = UNSUPPORTED
|
|
107
|
+
expert_count: MeasurementField = UNSUPPORTED
|
|
108
|
+
quantization: str | None = None
|
|
109
|
+
weight_format: str | None = None
|
|
110
|
+
size_bytes: MeasurementField = UNSUPPORTED
|
|
111
|
+
max_context: MeasurementField = UNSUPPORTED
|
|
112
|
+
embedding_dim: MeasurementField = UNSUPPORTED
|
|
113
|
+
layers: MeasurementField = UNSUPPORTED
|
|
114
|
+
attention_heads: MeasurementField = UNSUPPORTED
|
|
115
|
+
kv_heads: MeasurementField = UNSUPPORTED
|
|
116
|
+
head_dim: MeasurementField = UNSUPPORTED
|
|
117
|
+
vocab_size: MeasurementField = UNSUPPORTED
|
|
118
|
+
rope_config: dict[str, Any] | None = None
|
|
119
|
+
sliding_window: MeasurementField = UNSUPPORTED
|
|
120
|
+
declared_capabilities: WireSequence[WireEnum[ModelCapabilityFlag]] = ()
|
|
121
|
+
license_text: str | None = None
|
|
122
|
+
|
|
123
|
+
@model_validator(mode="after")
|
|
124
|
+
def _check_identity_coherence(self) -> Self:
|
|
125
|
+
"""Recompute the identity triple's derived fields and require them to agree.
|
|
126
|
+
|
|
127
|
+
Unlike a machine fingerprint — whose inclusion policy is allowed to change under a
|
|
128
|
+
profile that must still reconstruct exactly as it was written — a canonical ID and an
|
|
129
|
+
identity confidence are pure functions of the triple with no such historical caveat
|
|
130
|
+
(:class:`baseaicore.ModelIdentity` never re-verifies a stored fingerprint for that reason,
|
|
131
|
+
but always recomputes ``canonical_id``). Recomputing here is therefore safe, not merely
|
|
132
|
+
convenient, and it catches a producer that materialized the derived fields inconsistently
|
|
133
|
+
with the triple that stands next to them.
|
|
134
|
+
|
|
135
|
+
Raises:
|
|
136
|
+
ValueError: If ``provider_model_name`` or ``artifact_digest`` fails
|
|
137
|
+
:class:`baseaicore.ModelIdentity`'s own validation, or if ``canonical_id`` or
|
|
138
|
+
``identity_confidence`` disagrees with what the triple recomputes.
|
|
139
|
+
"""
|
|
140
|
+
try:
|
|
141
|
+
identity = ModelIdentity(
|
|
142
|
+
provider_kind=self.provider_kind,
|
|
143
|
+
provider_model_name=self.provider_model_name,
|
|
144
|
+
artifact_digest=self.artifact_digest,
|
|
145
|
+
)
|
|
146
|
+
except SuiteValidationError as exc:
|
|
147
|
+
# baseaicore.ValidationError is a SuiteError, not a ValueError; pydantic only
|
|
148
|
+
# aggregates ValueError/AssertionError into its own report (setspec.serialization
|
|
149
|
+
# hits the same seam for timestamps).
|
|
150
|
+
raise ValueError(str(exc)) from exc
|
|
151
|
+
if identity.canonical_id != self.canonical_id:
|
|
152
|
+
raise ValueError(
|
|
153
|
+
f"canonical_id {self.canonical_id!r} does not match the identity triple, which "
|
|
154
|
+
f"recomputes to {identity.canonical_id!r}. canonical_id is a pure function of "
|
|
155
|
+
"provider_kind, provider_model_name and artifact_digest (ADR-0024) — it is "
|
|
156
|
+
"carried on the wire for convenience, not as an independent fact."
|
|
157
|
+
)
|
|
158
|
+
if identity.identity_confidence.value != self.identity_confidence:
|
|
159
|
+
raise ValueError(
|
|
160
|
+
f"identity_confidence {self.identity_confidence!r} does not match the identity "
|
|
161
|
+
f"triple: artifact_digest is "
|
|
162
|
+
f"{'present' if self.artifact_digest is not None else 'absent'}, which makes "
|
|
163
|
+
f"this identity {identity.identity_confidence.value!r}."
|
|
164
|
+
)
|
|
165
|
+
return self
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
ModelIdentityOut, ModelIdentityIn = payload_models(ModelIdentityFields)
|
|
169
|
+
"""The ``model.identity`` payload pair: ``Out`` for writers, ``In`` for readers."""
|
setspec/provenance.py
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""Contract module — provenance building blocks shared by more than one versioned payload.
|
|
2
|
+
|
|
3
|
+
Imports pydantic and :mod:`baseaicore`; performs no I/O.
|
|
4
|
+
|
|
5
|
+
This module exists for the same reason :mod:`setspec.metrics` does. A sub-model used by two
|
|
6
|
+
payload types belongs to neither of them: if ``EnvironmentFields`` lived in
|
|
7
|
+
:mod:`setspec.benchmark.v1`, then the day ``benchmark.result`` needs a changed environment shape,
|
|
8
|
+
whoever edits that class would silently change ``capability.evidence`` v1 too — a frozen contract
|
|
9
|
+
mutating because someone edited a different payload's module, which is precisely the drift
|
|
10
|
+
ADR-0009 exists to prevent. A shared block gets
|
|
11
|
+
a neutral home so that changing it is an obviously cross-cutting act rather than an accident.
|
|
12
|
+
|
|
13
|
+
Nothing here is a wire payload in its own right: none of these names appears in
|
|
14
|
+
:data:`~setspec.envelope.SUPPORTED_SCHEMAS`, and none is generated into an ``Out``/``In`` pair.
|
|
15
|
+
They are field groups that versioned payloads embed, which is why they carry no version of their
|
|
16
|
+
own — their version is whichever payload's version they appear inside.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from baseaicore import ProviderKind
|
|
22
|
+
from pydantic import Field
|
|
23
|
+
|
|
24
|
+
from setspec.base import PayloadDefinition, WireEnum
|
|
25
|
+
|
|
26
|
+
__all__ = ["EnvironmentFields"]
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class EnvironmentFields(PayloadDefinition):
|
|
30
|
+
"""Provider and drift-sensitive environment facts at the moment of measurement.
|
|
31
|
+
|
|
32
|
+
Unifies two bullets that
|
|
33
|
+
Machine Identity §6 and
|
|
34
|
+
ADR-0022 §1 describe separately
|
|
35
|
+
but with identical content — "provider kind + provider version" and "GPU driver, CUDA, OS
|
|
36
|
+
version at measurement" — into the one nested object ADR-0022 already names ``environment`` on
|
|
37
|
+
``capability.evidence``, so a benchmark result and the evidence aggregated from it carry
|
|
38
|
+
environment facts in the same shape rather than two shapes a consumer has to reconcile.
|
|
39
|
+
|
|
40
|
+
Every field but the provider's own identity is a **drift signal, not identity**: a driver
|
|
41
|
+
upgrade or an OS patch must never re-identify a machine (that is the machine fingerprint's
|
|
42
|
+
job, and it deliberately excludes all of these), but it does reduce confidence in performance
|
|
43
|
+
evidence measured before it
|
|
44
|
+
(ADR-0017's
|
|
45
|
+
``environment_factor``). Recording them is what makes that reduction computable at all.
|
|
46
|
+
|
|
47
|
+
Attributes:
|
|
48
|
+
provider_kind: Which kind of provider served the model for this measurement.
|
|
49
|
+
provider_version: The provider's own version string, e.g. ``"0.32.13"``.
|
|
50
|
+
gpu_driver_version: A drift signal; ``None`` when no GPU was involved.
|
|
51
|
+
cuda_version: The CUDA/ROCm toolkit version. A drift signal.
|
|
52
|
+
os_version: A drift signal.
|
|
53
|
+
"""
|
|
54
|
+
|
|
55
|
+
provider_kind: WireEnum[ProviderKind]
|
|
56
|
+
provider_version: str = Field(min_length=1)
|
|
57
|
+
gpu_driver_version: str | None = None
|
|
58
|
+
cuda_version: str | None = None
|
|
59
|
+
os_version: str | None = None
|
setspec/py.typed
ADDED
|
File without changes
|
setspec/schemas/.gitkeep
ADDED
|
File without changes
|