setspec 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- setspec/__about__.py +3 -0
- setspec/__init__.py +114 -0
- setspec/artifacts.py +4 -0
- setspec/base.py +256 -0
- setspec/benchmark/v1.py +594 -0
- setspec/capability/v1.py +394 -0
- setspec/envelope.py +468 -0
- setspec/error/v1.py +4 -0
- setspec/errors.py +68 -0
- setspec/event/v1.py +4 -0
- setspec/goal/v1.py +390 -0
- setspec/goldens/.gitkeep +0 -0
- setspec/machine/v1.py +131 -0
- setspec/metrics.py +184 -0
- setspec/model/v1.py +169 -0
- setspec/provenance.py +59 -0
- setspec/py.typed +0 -0
- setspec/schemas/.gitkeep +0 -0
- setspec/serialization.py +330 -0
- setspec/vocabulary.py +205 -0
- setspec-0.2.0.dist-info/METADATA +160 -0
- setspec-0.2.0.dist-info/RECORD +24 -0
- setspec-0.2.0.dist-info/WHEEL +4 -0
- setspec-0.2.0.dist-info/licenses/LICENSE +201 -0
setspec/benchmark/v1.py
ADDED
|
@@ -0,0 +1,594 @@
|
|
|
1
|
+
"""Contract module — ``benchmark.result`` and ``benchmark.run_summary`` v1.
|
|
2
|
+
|
|
3
|
+
Imports pydantic and :mod:`baseaicore`; performs no I/O. ``BenchmarkResult`` is one benchmark, one
|
|
4
|
+
measurement subject, metrics plus provenance plus a samples reference;
|
|
5
|
+
``BenchmarkRunSummary`` is one run: subject, suite, status, timings, aggregate metrics
|
|
6
|
+
([spec §7](../../../docs/packages/setspec/spec.md)).
|
|
7
|
+
|
|
8
|
+
**Status: draft (`1.0`).** See :mod:`setspec.model.v1` for what that means. The known risk named
|
|
9
|
+
by [development plan Phase 2](../../../docs/packages/setspec/development-plan.md) is guessing the
|
|
10
|
+
result shape before FreeWeight exists to produce one; this module is built from the one place that
|
|
11
|
+
shape is already normative before any FreeWeight code exists —
|
|
12
|
+
Machine Identity §6, "what
|
|
13
|
+
every measured result must carry" — plus
|
|
14
|
+
ADR-0022 and
|
|
15
|
+
ADR-0023 for the fields those provenance
|
|
16
|
+
bullets expand into. A field with no normative source here is a field this module does
|
|
17
|
+
not invent; Phase 4 corrects any gap against FreeWeight's real output.
|
|
18
|
+
|
|
19
|
+
**Hash-shaped fields are checked, never guessed, and never both.** Two kinds of string on this
|
|
20
|
+
result claim to be a hash of something: ``runtime_profile_hash`` is recomputed from the embedded
|
|
21
|
+
``runtime_profile`` and compared, the same reasoning :mod:`setspec.model.v1` applies to
|
|
22
|
+
``canonical_id`` — it is a pure function with no real-world format ambiguity. Every other
|
|
23
|
+
hash-shaped field (``manifest_hash``, ``prompt_subset_hash``, ``reproducibility_fingerprint``, each
|
|
24
|
+
entry of ``dataset_hashes``) is validated only as a non-empty string: this package cannot yet
|
|
25
|
+
compute FreeWeight's own manifest or fingerprint hashes, and guessing a format — hex-only, with or
|
|
26
|
+
without an algorithm prefix — risks rejecting the first real result over a formatting nuance
|
|
27
|
+
instead of the risk the tests actually target.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
from enum import StrEnum
|
|
33
|
+
from typing import Any, Self
|
|
34
|
+
|
|
35
|
+
from baseaicore import UNSUPPORTED, RuntimeProfile
|
|
36
|
+
from pydantic import Field, model_validator
|
|
37
|
+
|
|
38
|
+
from setspec.base import PayloadDefinition, WireEnum, WireSequence, payload_models
|
|
39
|
+
from setspec.machine.v1 import MachineProfileFields
|
|
40
|
+
from setspec.metrics import MetricValueFields
|
|
41
|
+
from setspec.model.v1 import ModelIdentityFields
|
|
42
|
+
from setspec.provenance import EnvironmentFields
|
|
43
|
+
from setspec.serialization import MeasurementField, TimestampField
|
|
44
|
+
|
|
45
|
+
__all__ = [
|
|
46
|
+
"ApplicationProvenanceFields",
|
|
47
|
+
"BenchmarkResultFields",
|
|
48
|
+
"BenchmarkResultIn",
|
|
49
|
+
"BenchmarkResultOut",
|
|
50
|
+
"BenchmarkResultStatus",
|
|
51
|
+
"BenchmarkRunSummaryFields",
|
|
52
|
+
"BenchmarkRunSummaryIn",
|
|
53
|
+
"BenchmarkRunSummaryOut",
|
|
54
|
+
"BenchmarkSuiteProvenanceFields",
|
|
55
|
+
"ExecutionProvenanceFields",
|
|
56
|
+
"MeasurementClass",
|
|
57
|
+
"PromptUsageFields",
|
|
58
|
+
"ReproducibilityFingerprintFields",
|
|
59
|
+
"RunStatus",
|
|
60
|
+
"RuntimeProfileFields",
|
|
61
|
+
"ServedContextSource",
|
|
62
|
+
"TelemetrySummaryFields",
|
|
63
|
+
]
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class RuntimeProfileFields(PayloadDefinition):
|
|
67
|
+
"""How a provider was asked to load and serve the model — nested only, never enveloped alone.
|
|
68
|
+
|
|
69
|
+
Mirrors :class:`baseaicore.RuntimeProfile` field for field, including that every field is
|
|
70
|
+
optional: a profile with everything unset means "provider defaults" and is itself a legal,
|
|
71
|
+
hashable profile (ADR-0023 §1).
|
|
72
|
+
|
|
73
|
+
Attributes:
|
|
74
|
+
context_size: Requested context window, in tokens.
|
|
75
|
+
kv_cache_precision: KV-cache quantization, e.g. ``"f16"``, ``"q8_0"``.
|
|
76
|
+
gpu_layers: Number of layers offloaded to GPU.
|
|
77
|
+
flash_attention: Whether flash attention was requested.
|
|
78
|
+
threads: CPU thread count requested.
|
|
79
|
+
batch_size: Requested batch size.
|
|
80
|
+
keep_alive: How long the provider was asked to keep the model loaded, e.g. ``"5m"``.
|
|
81
|
+
provider_options: Anything provider-specific with no field of its own.
|
|
82
|
+
"""
|
|
83
|
+
|
|
84
|
+
context_size: int | None = None
|
|
85
|
+
kv_cache_precision: str | None = None
|
|
86
|
+
gpu_layers: int | None = None
|
|
87
|
+
flash_attention: bool | None = None
|
|
88
|
+
threads: int | None = None
|
|
89
|
+
batch_size: int | None = None
|
|
90
|
+
keep_alive: str | None = None
|
|
91
|
+
provider_options: dict[str, Any] = Field(default_factory=dict)
|
|
92
|
+
|
|
93
|
+
@property
|
|
94
|
+
def profile_hash(self) -> str:
|
|
95
|
+
"""Return this profile's 16-character hash, computed by the domain type that defines it.
|
|
96
|
+
|
|
97
|
+
Public because a consumer needs it for the same reason a producer does: two results are
|
|
98
|
+
comparable only if their profiles hash identically
|
|
99
|
+
(ADR-0023), and a reader holding a
|
|
100
|
+
payload should not have to reconstruct a :class:`baseaicore.RuntimeProfile` by hand to
|
|
101
|
+
find that out. Delegating rather than reimplementing is the point — a second
|
|
102
|
+
implementation of this hash would eventually disagree with the first, and the whole value
|
|
103
|
+
of the hash is that it does not.
|
|
104
|
+
|
|
105
|
+
Returns:
|
|
106
|
+
16 lowercase hex characters, identical to what
|
|
107
|
+
:attr:`baseaicore.RuntimeProfile.profile_hash` computes for the same field values.
|
|
108
|
+
"""
|
|
109
|
+
return RuntimeProfile(
|
|
110
|
+
context_size=self.context_size,
|
|
111
|
+
kv_cache_precision=self.kv_cache_precision,
|
|
112
|
+
gpu_layers=self.gpu_layers,
|
|
113
|
+
flash_attention=self.flash_attention,
|
|
114
|
+
threads=self.threads,
|
|
115
|
+
batch_size=self.batch_size,
|
|
116
|
+
keep_alive=self.keep_alive,
|
|
117
|
+
provider_options=dict(self.provider_options),
|
|
118
|
+
).profile_hash
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
class ApplicationProvenanceFields(PayloadDefinition):
|
|
122
|
+
"""Which build of the producing application measured this result.
|
|
123
|
+
|
|
124
|
+
All three fields are part of Machine Identity §6's minimum provenance set — none is optional
|
|
125
|
+
enrichment — because "which code produced this number" is exactly what a regression hunt needs
|
|
126
|
+
first.
|
|
127
|
+
|
|
128
|
+
Attributes:
|
|
129
|
+
name: The producing application's distribution name, e.g. ``"freeweight"``.
|
|
130
|
+
version: That application's version string.
|
|
131
|
+
git_commit: The commit the running build was checked out at.
|
|
132
|
+
"""
|
|
133
|
+
|
|
134
|
+
name: str = Field(min_length=1)
|
|
135
|
+
version: str = Field(min_length=1)
|
|
136
|
+
git_commit: str = Field(min_length=1)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class PromptUsageFields(PayloadDefinition):
|
|
140
|
+
"""One prompt this benchmark used, identified precisely enough to reproduce it.
|
|
141
|
+
|
|
142
|
+
Attributes:
|
|
143
|
+
prompt_id: The prompt's identifier in its manifest.
|
|
144
|
+
version: The prompt's own version.
|
|
145
|
+
sha256: The prompt's content hash, as computed by whatever produced it — not validated
|
|
146
|
+
against a fixed hex format here; see the module docstring's note on hash-shaped
|
|
147
|
+
fields.
|
|
148
|
+
"""
|
|
149
|
+
|
|
150
|
+
prompt_id: str = Field(min_length=1)
|
|
151
|
+
version: str = Field(min_length=1)
|
|
152
|
+
sha256: str = Field(min_length=1)
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
class BenchmarkSuiteProvenanceFields(PayloadDefinition):
|
|
156
|
+
"""Which benchmark suite produced this result, and at what version.
|
|
157
|
+
|
|
158
|
+
A differing ``suite_version``, ``dataset_hashes`` entry or ``prompt_subset_hash`` is a hard
|
|
159
|
+
separation for confidence purposes, never a discount
|
|
160
|
+
(ADR-0017) — SetSpec carries
|
|
161
|
+
these values; the separation itself is LoadCoach's and FreeWeight's behaviour to apply.
|
|
162
|
+
|
|
163
|
+
Attributes:
|
|
164
|
+
suite_key: The suite's identifier, e.g. ``"native.tool_use"``.
|
|
165
|
+
suite_version: The suite's own version. A version bump separates results from different
|
|
166
|
+
versions; they are never averaged together.
|
|
167
|
+
category: The suite's category, e.g. ``"tool_use"``.
|
|
168
|
+
runner: How the suite executes, e.g. ``"native"`` or ``"external"``.
|
|
169
|
+
manifest_hash: Hash of the suite's manifest, as the producer computed it.
|
|
170
|
+
dataset_hashes: Hash per named dataset the suite depends on; empty when the suite uses
|
|
171
|
+
none.
|
|
172
|
+
prompt_subset_hash: Hash of only the prompts *this suite* declares — the fingerprint
|
|
173
|
+
input, per benchmark and not per pack
|
|
174
|
+
(ADR-0028).
|
|
175
|
+
prompts_used: Every prompt named by ``prompt_subset_hash``, individually identified.
|
|
176
|
+
"""
|
|
177
|
+
|
|
178
|
+
suite_key: str = Field(min_length=1)
|
|
179
|
+
suite_version: str = Field(min_length=1)
|
|
180
|
+
category: str | None = None
|
|
181
|
+
runner: str | None = None
|
|
182
|
+
manifest_hash: str = Field(min_length=1)
|
|
183
|
+
dataset_hashes: dict[str, str] = Field(default_factory=dict)
|
|
184
|
+
prompt_subset_hash: str = Field(min_length=1)
|
|
185
|
+
prompts_used: WireSequence[PromptUsageFields] = ()
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
class ServedContextSource(StrEnum):
|
|
189
|
+
"""Where a run's actually-served context length came from.
|
|
190
|
+
|
|
191
|
+
Distinct from :attr:`~setspec.model.v1.ModelIdentityFields.max_context`, which is what the
|
|
192
|
+
model *advertises*, not what a provider was actually configured to serve
|
|
193
|
+
(ADR-0023 §4).
|
|
194
|
+
"""
|
|
195
|
+
|
|
196
|
+
CONFIGURED = "configured"
|
|
197
|
+
"""An operator or the runtime profile explicitly set it."""
|
|
198
|
+
|
|
199
|
+
REPORTED = "reported"
|
|
200
|
+
"""The provider reported the context it actually served."""
|
|
201
|
+
|
|
202
|
+
ASSUMED = "assumed"
|
|
203
|
+
"""Neither configured nor reported; taken from the model's advertised default."""
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
class ExecutionProvenanceFields(PayloadDefinition):
|
|
207
|
+
"""The execution parameters actually in effect for this result, after precedence resolution.
|
|
208
|
+
|
|
209
|
+
Attributes:
|
|
210
|
+
effective_parameters: Resolved sampling and limit parameters, post-precedence-chain —
|
|
211
|
+
what was actually used, not what any one layer requested.
|
|
212
|
+
repetitions: How many repetitions this result's configuration requested.
|
|
213
|
+
sample_count: How many samples this result actually produced.
|
|
214
|
+
seed: The seed used, or the literal string ``"nondeterministic"`` when none was set.
|
|
215
|
+
served_context: The context length actually served.
|
|
216
|
+
served_context_source: Where that value came from.
|
|
217
|
+
gpu_index: The device this result's metrics are attributed to.
|
|
218
|
+
multi_gpu_visible: Whether more than one GPU was visible during measurement
|
|
219
|
+
(ADR-0027).
|
|
220
|
+
"""
|
|
221
|
+
|
|
222
|
+
effective_parameters: dict[str, Any] = Field(default_factory=dict)
|
|
223
|
+
repetitions: int = Field(ge=1)
|
|
224
|
+
sample_count: int = Field(ge=0)
|
|
225
|
+
seed: int | str
|
|
226
|
+
served_context: int = Field(ge=0)
|
|
227
|
+
served_context_source: WireEnum[ServedContextSource]
|
|
228
|
+
gpu_index: int = Field(ge=0, default=0)
|
|
229
|
+
multi_gpu_visible: bool = False
|
|
230
|
+
|
|
231
|
+
@model_validator(mode="after")
|
|
232
|
+
def _check_seed(self) -> Self:
|
|
233
|
+
"""Require a string seed to be exactly the documented sentinel.
|
|
234
|
+
|
|
235
|
+
Raises:
|
|
236
|
+
ValueError: If ``seed`` is a string other than ``"nondeterministic"`` — a numeric
|
|
237
|
+
string masquerading as a seed is a producer bug the schema should catch, not
|
|
238
|
+
silently coerce.
|
|
239
|
+
"""
|
|
240
|
+
if isinstance(self.seed, str) and self.seed != "nondeterministic":
|
|
241
|
+
raise ValueError(
|
|
242
|
+
f"seed must be an integer or the literal string 'nondeterministic'; got "
|
|
243
|
+
f"{self.seed!r}. A run that did not set a seed says so with the sentinel, not "
|
|
244
|
+
"with any other string."
|
|
245
|
+
)
|
|
246
|
+
return self
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
class ReproducibilityFingerprintFields(PayloadDefinition):
|
|
250
|
+
"""The answer to "could this measurement be repeated, and is that other result the same
|
|
251
|
+
thing?"
|
|
252
|
+
(Machine Identity §4).
|
|
253
|
+
|
|
254
|
+
Attributes:
|
|
255
|
+
reproducibility_fingerprint: The hash itself, as the producer computed it.
|
|
256
|
+
fingerprint_document: The **full input document that was hashed**, stored verbatim per
|
|
257
|
+
Machine Identity §4 rule 2 — "a hash you cannot explain is useless during a regression
|
|
258
|
+
hunt." Kept as a structured but untyped mapping rather than a fully modeled nested
|
|
259
|
+
structure: its shape mirrors provenance already typed elsewhere on this result, and a
|
|
260
|
+
second, independently-typed copy of that shape would drift from the first the moment
|
|
261
|
+
either one changed.
|
|
262
|
+
"""
|
|
263
|
+
|
|
264
|
+
reproducibility_fingerprint: str = Field(min_length=1)
|
|
265
|
+
fingerprint_document: dict[str, Any]
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
class MeasurementClass(StrEnum):
|
|
269
|
+
"""Whether the model was cold, warm, or serving a reused cache when this was measured.
|
|
270
|
+
|
|
271
|
+
The cold/warm marker
|
|
272
|
+
Machine Identity §6
|
|
273
|
+
calls "optional but strongly recommended". For a load-time or first-token metric it is not
|
|
274
|
+
optional in practice: the same benchmark against the same weights reports figures that differ
|
|
275
|
+
by an order of magnitude depending on this value alone, so a result that omits it is a number
|
|
276
|
+
two readers will interpret differently.
|
|
277
|
+
"""
|
|
278
|
+
|
|
279
|
+
COLD = "cold"
|
|
280
|
+
"""Measured on a fresh load, with nothing cached."""
|
|
281
|
+
|
|
282
|
+
WARM = "warm"
|
|
283
|
+
"""Measured with the model already resident."""
|
|
284
|
+
|
|
285
|
+
CACHE_REUSED = "cache_reused"
|
|
286
|
+
"""Measured while a prompt or KV cache from an earlier request was still in play."""
|
|
287
|
+
|
|
288
|
+
NOT_APPLICABLE = "n/a"
|
|
289
|
+
"""The distinction does not apply to what this benchmark measures."""
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
class TelemetrySummaryFields(PayloadDefinition):
|
|
293
|
+
"""What the machine was doing while this benchmark ran, reduced to the figures that explain it.
|
|
294
|
+
|
|
295
|
+
The telemetry summary
|
|
296
|
+
Machine Identity §6
|
|
297
|
+
names as "required for any benchmark whose numbers depend on them" — a memory or energy
|
|
298
|
+
benchmark's result is not interpretable without it, and a throttled run's timings mean
|
|
299
|
+
something different from an unthrottled one's.
|
|
300
|
+
|
|
301
|
+
Every figure is device-scoped by the enclosing result's ``execution.gpu_index``, never
|
|
302
|
+
machine-wide: there is no meaningful sum of two GPUs' VRAM or power, and
|
|
303
|
+
ADR-0027 requires each derived figure to name
|
|
304
|
+
the device it came from rather than silently aggregating across a machine the reference
|
|
305
|
+
hardware does not have.
|
|
306
|
+
|
|
307
|
+
Attributes:
|
|
308
|
+
peak_vram_bytes: Highest device memory in use during the benchmark.
|
|
309
|
+
peak_power_watts: Highest instantaneous draw observed.
|
|
310
|
+
mean_power_watts: Mean draw across the benchmark — the figure an energy estimate derives
|
|
311
|
+
from, and never interchangeable with the peak.
|
|
312
|
+
max_temperature_c: Highest device temperature observed.
|
|
313
|
+
throttled: Whether the device throttled at any point. ``None`` when the driver exposes no
|
|
314
|
+
throttle-reason mask to read, which is a different statement from ``False``: guessing
|
|
315
|
+
"not throttled" would attribute a slow result to the model rather than to the hardware
|
|
316
|
+
protecting itself.
|
|
317
|
+
"""
|
|
318
|
+
|
|
319
|
+
peak_vram_bytes: MeasurementField = UNSUPPORTED
|
|
320
|
+
peak_power_watts: MeasurementField = UNSUPPORTED
|
|
321
|
+
mean_power_watts: MeasurementField = UNSUPPORTED
|
|
322
|
+
max_temperature_c: MeasurementField = UNSUPPORTED
|
|
323
|
+
throttled: bool | None = None
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
class BenchmarkResultStatus(StrEnum):
|
|
327
|
+
"""Terminal states of one benchmark within a run (``run_tests.status`` in FreeWeight's data
|
|
328
|
+
model). A failed benchmark never fails its run; a failed sample never fails its benchmark."""
|
|
329
|
+
|
|
330
|
+
COMPLETED = "completed"
|
|
331
|
+
FAILED = "failed"
|
|
332
|
+
SKIPPED = "skipped"
|
|
333
|
+
CANCELLED = "cancelled"
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
class BenchmarkResultFields(PayloadDefinition):
|
|
337
|
+
"""Field definitions for ``benchmark.result``; use :data:`BenchmarkResultOut` /
|
|
338
|
+
:data:`BenchmarkResultIn`.
|
|
339
|
+
|
|
340
|
+
One benchmark, one measurement subject, metrics plus provenance plus a samples reference
|
|
341
|
+
([spec §7](../../../docs/packages/setspec/spec.md)). The provenance fields are exactly Machine
|
|
342
|
+
Identity §6's minimum set — see the module docstring for how each bullet there maps onto a
|
|
343
|
+
field or nested object here.
|
|
344
|
+
|
|
345
|
+
Attributes:
|
|
346
|
+
model: The measured weights, plus what the provider reports about them.
|
|
347
|
+
runtime_profile: How the provider was asked to serve the model for this measurement.
|
|
348
|
+
runtime_profile_hash: :attr:`runtime_profile`'s hash — checked, not merely carried; see
|
|
349
|
+
:meth:`_check_runtime_profile_hash`.
|
|
350
|
+
machine_fingerprint: Where this was measured.
|
|
351
|
+
machine_profile: A full snapshot of the machine at measurement time, when the producer
|
|
352
|
+
chose to embed one rather than carry only the fingerprint.
|
|
353
|
+
suite: Which benchmark suite produced this, and at what version.
|
|
354
|
+
execution: The execution parameters actually in effect.
|
|
355
|
+
environment: Provider and drift-sensitive facts at measurement time.
|
|
356
|
+
application: Which build of the producing application measured this.
|
|
357
|
+
reproducibility: The fingerprint proving this measurement could be repeated.
|
|
358
|
+
started_at: When measurement began.
|
|
359
|
+
completed_at: When measurement ended; never earlier than ``started_at``.
|
|
360
|
+
status: This benchmark's terminal state.
|
|
361
|
+
skip_reason: Present iff ``status`` is ``skipped``.
|
|
362
|
+
error_code: A stable, machine-readable reason this benchmark did not complete. Optional
|
|
363
|
+
because not every non-completion has one — a cancelled benchmark was simply stopped —
|
|
364
|
+
but a ``failed`` result that omits it is a failure nobody can triage.
|
|
365
|
+
error_text: Human-readable detail alongside ``error_code``.
|
|
366
|
+
completed_cases: How many of this benchmark's cases finished.
|
|
367
|
+
total_cases: How many cases this benchmark set out to run. Never smaller than
|
|
368
|
+
``completed_cases`` — see :meth:`_check_case_counts`.
|
|
369
|
+
measurement_class: Whether this was measured cold, warm, or against a reused cache.
|
|
370
|
+
telemetry_summary: What the hardware was doing while this ran.
|
|
371
|
+
raw_response_ref: An opaque, producer-local pointer to the stored raw provider response.
|
|
372
|
+
metrics: Every metric this benchmark measured. Empty for a skipped or cancelled result;
|
|
373
|
+
a completed result with none is rejected — see :meth:`_check_status_coherence`.
|
|
374
|
+
samples_ref: An opaque, producer-local pointer to the raw sample rows behind ``metrics``.
|
|
375
|
+
Never resolved by a consumer; carried only for the producer's own drill-down.
|
|
376
|
+
|
|
377
|
+
The last six are the provenance
|
|
378
|
+
Machine Identity §6
|
|
379
|
+
calls "optional but strongly recommended, and required for any benchmark whose numbers depend
|
|
380
|
+
on them", plus the two error fields FreeWeight's own ``run_tests`` row carries. They are
|
|
381
|
+
declared rather than left out because the writer half of this pair forbids unknown keys: a
|
|
382
|
+
field this schema does not name is a field a producer *cannot emit at all*, so omitting an
|
|
383
|
+
optional field is not a neutral act here — it is a decision that the field may never be sent.
|
|
384
|
+
"""
|
|
385
|
+
|
|
386
|
+
model: ModelIdentityFields
|
|
387
|
+
runtime_profile: RuntimeProfileFields
|
|
388
|
+
runtime_profile_hash: str = Field(min_length=1)
|
|
389
|
+
machine_fingerprint: str = Field(min_length=1)
|
|
390
|
+
machine_profile: MachineProfileFields | None = None
|
|
391
|
+
suite: BenchmarkSuiteProvenanceFields
|
|
392
|
+
execution: ExecutionProvenanceFields
|
|
393
|
+
environment: EnvironmentFields
|
|
394
|
+
application: ApplicationProvenanceFields
|
|
395
|
+
reproducibility: ReproducibilityFingerprintFields
|
|
396
|
+
started_at: TimestampField
|
|
397
|
+
completed_at: TimestampField
|
|
398
|
+
status: WireEnum[BenchmarkResultStatus]
|
|
399
|
+
skip_reason: str | None = None
|
|
400
|
+
error_code: str | None = None
|
|
401
|
+
error_text: str | None = None
|
|
402
|
+
completed_cases: int | None = Field(default=None, ge=0)
|
|
403
|
+
total_cases: int | None = Field(default=None, ge=0)
|
|
404
|
+
measurement_class: WireEnum[MeasurementClass] | None = None
|
|
405
|
+
telemetry_summary: TelemetrySummaryFields | None = None
|
|
406
|
+
raw_response_ref: str | None = None
|
|
407
|
+
metrics: WireSequence[MetricValueFields] = ()
|
|
408
|
+
samples_ref: str | None = None
|
|
409
|
+
|
|
410
|
+
@model_validator(mode="after")
|
|
411
|
+
def _check_runtime_profile_hash(self) -> Self:
|
|
412
|
+
"""Recompute ``runtime_profile_hash`` from the embedded profile and require agreement.
|
|
413
|
+
|
|
414
|
+
Safe to recompute for the same reason ``canonical_id`` is: a runtime profile's hash is a
|
|
415
|
+
pure function of its own fields with no historical-policy caveat, unlike a machine
|
|
416
|
+
fingerprint.
|
|
417
|
+
|
|
418
|
+
Raises:
|
|
419
|
+
ValueError: If the declared hash does not match what ``runtime_profile`` recomputes.
|
|
420
|
+
"""
|
|
421
|
+
recomputed = self.runtime_profile.profile_hash
|
|
422
|
+
if recomputed != self.runtime_profile_hash:
|
|
423
|
+
raise ValueError(
|
|
424
|
+
f"runtime_profile_hash {self.runtime_profile_hash!r} does not match "
|
|
425
|
+
f"runtime_profile, which recomputes to {recomputed!r}. The hash is a pure "
|
|
426
|
+
"function of the profile's own fields (ADR-0023) and is carried on the wire for "
|
|
427
|
+
"convenience, not as an independent fact."
|
|
428
|
+
)
|
|
429
|
+
return self
|
|
430
|
+
|
|
431
|
+
@model_validator(mode="after")
|
|
432
|
+
def _check_timing_order(self) -> Self:
|
|
433
|
+
"""Require ``completed_at`` not to precede ``started_at``.
|
|
434
|
+
|
|
435
|
+
Raises:
|
|
436
|
+
ValueError: If the two timestamps are out of order.
|
|
437
|
+
"""
|
|
438
|
+
if self.completed_at < self.started_at:
|
|
439
|
+
raise ValueError(
|
|
440
|
+
f"completed_at ({self.completed_at.isoformat()}) precedes started_at "
|
|
441
|
+
f"({self.started_at.isoformat()}); a benchmark cannot finish before it started."
|
|
442
|
+
)
|
|
443
|
+
return self
|
|
444
|
+
|
|
445
|
+
@model_validator(mode="after")
|
|
446
|
+
def _check_case_counts(self) -> Self:
|
|
447
|
+
"""Require ``completed_cases`` not to exceed ``total_cases`` when both are present.
|
|
448
|
+
|
|
449
|
+
The same arithmetic honesty :meth:`_check_timing_order` applies to timestamps: "12 of 10
|
|
450
|
+
cases done" is not a progress report, it is a producer bug, and a consumer rendering it as
|
|
451
|
+
a percentage would show something above 100%.
|
|
452
|
+
|
|
453
|
+
Raises:
|
|
454
|
+
ValueError: If both counts are present and ``completed_cases`` exceeds
|
|
455
|
+
``total_cases``.
|
|
456
|
+
"""
|
|
457
|
+
if (
|
|
458
|
+
self.completed_cases is not None
|
|
459
|
+
and self.total_cases is not None
|
|
460
|
+
and self.completed_cases > self.total_cases
|
|
461
|
+
):
|
|
462
|
+
raise ValueError(
|
|
463
|
+
f"completed_cases ({self.completed_cases}) exceeds total_cases "
|
|
464
|
+
f"({self.total_cases}); a benchmark cannot complete more cases than it ran."
|
|
465
|
+
)
|
|
466
|
+
return self
|
|
467
|
+
|
|
468
|
+
@model_validator(mode="after")
|
|
469
|
+
def _check_status_coherence(self) -> Self:
|
|
470
|
+
"""Require ``skip_reason`` and ``metrics`` to agree with ``status``.
|
|
471
|
+
|
|
472
|
+
Raises:
|
|
473
|
+
ValueError: If ``skip_reason`` is set without ``status == "skipped"``, if a skipped
|
|
474
|
+
result carries no ``skip_reason``, or if a completed result reports no metrics at
|
|
475
|
+
all — a completed benchmark that measured nothing is not a completed benchmark.
|
|
476
|
+
"""
|
|
477
|
+
if self.status is BenchmarkResultStatus.SKIPPED and self.skip_reason is None:
|
|
478
|
+
raise ValueError("a skipped result must name its skip_reason")
|
|
479
|
+
if self.status is not BenchmarkResultStatus.SKIPPED and self.skip_reason is not None:
|
|
480
|
+
raise ValueError(
|
|
481
|
+
f"skip_reason is set to {self.skip_reason!r} but status is {self.status.value!r}, "
|
|
482
|
+
"not 'skipped'; skip_reason is only meaningful for a skipped result"
|
|
483
|
+
)
|
|
484
|
+
if self.status is BenchmarkResultStatus.COMPLETED and not self.metrics:
|
|
485
|
+
raise ValueError(
|
|
486
|
+
"status is 'completed' but metrics is empty; a completed benchmark that measured "
|
|
487
|
+
"nothing is not a completed benchmark"
|
|
488
|
+
)
|
|
489
|
+
return self
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
BenchmarkResultOut, BenchmarkResultIn = payload_models(BenchmarkResultFields)
|
|
493
|
+
"""The ``benchmark.result`` payload pair: ``Out`` for writers, ``In`` for readers."""
|
|
494
|
+
|
|
495
|
+
|
|
496
|
+
class RunStatus(StrEnum):
|
|
497
|
+
"""A run's state, mirroring FreeWeight's run state machine (data model §3).
|
|
498
|
+
|
|
499
|
+
``INTERRUPTED`` is distinct from ``FAILED``: it means the process died, is discovered at
|
|
500
|
+
startup recovery, and is resumable with completed benchmarks preserved — a run summary
|
|
501
|
+
carrying it is not reporting a failure, it is reporting an unfinished run.
|
|
502
|
+
"""
|
|
503
|
+
|
|
504
|
+
QUEUED = "queued"
|
|
505
|
+
PREPARING = "preparing"
|
|
506
|
+
WARMING = "warming"
|
|
507
|
+
RUNNING = "running"
|
|
508
|
+
COMPLETED = "completed"
|
|
509
|
+
FAILED = "failed"
|
|
510
|
+
CANCELLING = "cancelling"
|
|
511
|
+
CANCELLED = "cancelled"
|
|
512
|
+
INTERRUPTED = "interrupted"
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
class BenchmarkRunSummaryFields(PayloadDefinition):
|
|
516
|
+
"""Field definitions for ``benchmark.run_summary``; use :data:`BenchmarkRunSummaryOut` /
|
|
517
|
+
:data:`BenchmarkRunSummaryIn`.
|
|
518
|
+
|
|
519
|
+
One run: subject, suite, status, timings, aggregate metrics
|
|
520
|
+
([spec §7](../../../docs/packages/setspec/spec.md)) — deliberately lighter than
|
|
521
|
+
:class:`BenchmarkResultFields`, which carries one benchmark's full per-result provenance; a run
|
|
522
|
+
summary is the roll-up of many such results and reuses the same subject/suite/environment
|
|
523
|
+
building blocks rather than repeating a full provenance set per aggregate metric.
|
|
524
|
+
|
|
525
|
+
Attributes:
|
|
526
|
+
model: The measurement subject's model identity.
|
|
527
|
+
runtime_profile: The runtime profile every benchmark in this run measured under.
|
|
528
|
+
runtime_profile_hash: Checked against ``runtime_profile``, as on a result.
|
|
529
|
+
machine_fingerprint: Where this run executed.
|
|
530
|
+
suite: The suite this run executed.
|
|
531
|
+
environment: Provider and drift-sensitive facts for the run.
|
|
532
|
+
application: Which build of the producing application ran this.
|
|
533
|
+
reproducibility: The run-level reproducibility fingerprint.
|
|
534
|
+
status: The run's current or terminal state.
|
|
535
|
+
created_at: When the run was created (queued).
|
|
536
|
+
started_at: When execution began; ``None`` if the run never left ``queued``.
|
|
537
|
+
completed_at: When execution ended; ``None`` while still in progress.
|
|
538
|
+
aggregate_metrics: Run-level rolled-up metrics, distinct from any one benchmark's own.
|
|
539
|
+
error_code: Present iff the run ended in a state a caller should investigate.
|
|
540
|
+
error_text: Human-readable detail alongside ``error_code``.
|
|
541
|
+
"""
|
|
542
|
+
|
|
543
|
+
model: ModelIdentityFields
|
|
544
|
+
runtime_profile: RuntimeProfileFields
|
|
545
|
+
runtime_profile_hash: str = Field(min_length=1)
|
|
546
|
+
machine_fingerprint: str = Field(min_length=1)
|
|
547
|
+
suite: BenchmarkSuiteProvenanceFields
|
|
548
|
+
environment: EnvironmentFields
|
|
549
|
+
application: ApplicationProvenanceFields
|
|
550
|
+
reproducibility: ReproducibilityFingerprintFields
|
|
551
|
+
status: WireEnum[RunStatus]
|
|
552
|
+
created_at: TimestampField
|
|
553
|
+
started_at: TimestampField | None = None
|
|
554
|
+
completed_at: TimestampField | None = None
|
|
555
|
+
aggregate_metrics: WireSequence[MetricValueFields] = ()
|
|
556
|
+
error_code: str | None = None
|
|
557
|
+
error_text: str | None = None
|
|
558
|
+
|
|
559
|
+
@model_validator(mode="after")
|
|
560
|
+
def _check_runtime_profile_hash(self) -> Self:
|
|
561
|
+
"""Recompute ``runtime_profile_hash`` from the embedded profile and require agreement.
|
|
562
|
+
|
|
563
|
+
Raises:
|
|
564
|
+
ValueError: If the declared hash does not match what ``runtime_profile`` recomputes.
|
|
565
|
+
"""
|
|
566
|
+
recomputed = self.runtime_profile.profile_hash
|
|
567
|
+
if recomputed != self.runtime_profile_hash:
|
|
568
|
+
raise ValueError(
|
|
569
|
+
f"runtime_profile_hash {self.runtime_profile_hash!r} does not match "
|
|
570
|
+
f"runtime_profile, which recomputes to {recomputed!r}."
|
|
571
|
+
)
|
|
572
|
+
return self
|
|
573
|
+
|
|
574
|
+
@model_validator(mode="after")
|
|
575
|
+
def _check_timing_order(self) -> Self:
|
|
576
|
+
"""Require ``completed_at`` not to precede ``started_at`` when both are present.
|
|
577
|
+
|
|
578
|
+
Raises:
|
|
579
|
+
ValueError: If both timestamps are present and out of order.
|
|
580
|
+
"""
|
|
581
|
+
if (
|
|
582
|
+
self.started_at is not None
|
|
583
|
+
and self.completed_at is not None
|
|
584
|
+
and self.completed_at < self.started_at
|
|
585
|
+
):
|
|
586
|
+
raise ValueError(
|
|
587
|
+
f"completed_at ({self.completed_at.isoformat()}) precedes started_at "
|
|
588
|
+
f"({self.started_at.isoformat()})."
|
|
589
|
+
)
|
|
590
|
+
return self
|
|
591
|
+
|
|
592
|
+
|
|
593
|
+
BenchmarkRunSummaryOut, BenchmarkRunSummaryIn = payload_models(BenchmarkRunSummaryFields)
|
|
594
|
+
"""The ``benchmark.run_summary`` payload pair: ``Out`` for writers, ``In`` for readers."""
|