setspec 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,594 @@
1
+ """Contract module — ``benchmark.result`` and ``benchmark.run_summary`` v1.
2
+
3
+ Imports pydantic and :mod:`baseaicore`; performs no I/O. ``BenchmarkResult`` is one benchmark, one
4
+ measurement subject, metrics plus provenance plus a samples reference;
5
+ ``BenchmarkRunSummary`` is one run: subject, suite, status, timings, aggregate metrics
6
+ ([spec §7](../../../docs/packages/setspec/spec.md)).
7
+
8
+ **Status: draft (`1.0`).** See :mod:`setspec.model.v1` for what that means. The known risk named
9
+ by [development plan Phase 2](../../../docs/packages/setspec/development-plan.md) is guessing the
10
+ result shape before FreeWeight exists to produce one; this module is built from the one place that
11
+ shape is already normative before any FreeWeight code exists —
12
+ Machine Identity §6, "what
13
+ every measured result must carry" — plus
14
+ ADR-0022 and
15
+ ADR-0023 for the fields those provenance
16
+ bullets expand into. A field with no normative source here is a field this module does
17
+ not invent; Phase 4 corrects any gap against FreeWeight's real output.
18
+
19
+ **Hash-shaped fields are checked, never guessed, and never both.** Two kinds of string on this
20
+ result claim to be a hash of something: ``runtime_profile_hash`` is recomputed from the embedded
21
+ ``runtime_profile`` and compared, the same reasoning :mod:`setspec.model.v1` applies to
22
+ ``canonical_id`` — it is a pure function with no real-world format ambiguity. Every other
23
+ hash-shaped field (``manifest_hash``, ``prompt_subset_hash``, ``reproducibility_fingerprint``, each
24
+ entry of ``dataset_hashes``) is validated only as a non-empty string: this package cannot yet
25
+ compute FreeWeight's own manifest or fingerprint hashes, and guessing a format — hex-only, with or
26
+ without an algorithm prefix — risks rejecting the first real result over a formatting nuance
27
+ instead of the risk the tests actually target.
28
+ """
29
+
30
+ from __future__ import annotations
31
+
32
+ from enum import StrEnum
33
+ from typing import Any, Self
34
+
35
+ from baseaicore import UNSUPPORTED, RuntimeProfile
36
+ from pydantic import Field, model_validator
37
+
38
+ from setspec.base import PayloadDefinition, WireEnum, WireSequence, payload_models
39
+ from setspec.machine.v1 import MachineProfileFields
40
+ from setspec.metrics import MetricValueFields
41
+ from setspec.model.v1 import ModelIdentityFields
42
+ from setspec.provenance import EnvironmentFields
43
+ from setspec.serialization import MeasurementField, TimestampField
44
+
45
+ __all__ = [
46
+ "ApplicationProvenanceFields",
47
+ "BenchmarkResultFields",
48
+ "BenchmarkResultIn",
49
+ "BenchmarkResultOut",
50
+ "BenchmarkResultStatus",
51
+ "BenchmarkRunSummaryFields",
52
+ "BenchmarkRunSummaryIn",
53
+ "BenchmarkRunSummaryOut",
54
+ "BenchmarkSuiteProvenanceFields",
55
+ "ExecutionProvenanceFields",
56
+ "MeasurementClass",
57
+ "PromptUsageFields",
58
+ "ReproducibilityFingerprintFields",
59
+ "RunStatus",
60
+ "RuntimeProfileFields",
61
+ "ServedContextSource",
62
+ "TelemetrySummaryFields",
63
+ ]
64
+
65
+
66
+ class RuntimeProfileFields(PayloadDefinition):
67
+ """How a provider was asked to load and serve the model — nested only, never enveloped alone.
68
+
69
+ Mirrors :class:`baseaicore.RuntimeProfile` field for field, including that every field is
70
+ optional: a profile with everything unset means "provider defaults" and is itself a legal,
71
+ hashable profile (ADR-0023 §1).
72
+
73
+ Attributes:
74
+ context_size: Requested context window, in tokens.
75
+ kv_cache_precision: KV-cache quantization, e.g. ``"f16"``, ``"q8_0"``.
76
+ gpu_layers: Number of layers offloaded to GPU.
77
+ flash_attention: Whether flash attention was requested.
78
+ threads: CPU thread count requested.
79
+ batch_size: Requested batch size.
80
+ keep_alive: How long the provider was asked to keep the model loaded, e.g. ``"5m"``.
81
+ provider_options: Anything provider-specific with no field of its own.
82
+ """
83
+
84
+ context_size: int | None = None
85
+ kv_cache_precision: str | None = None
86
+ gpu_layers: int | None = None
87
+ flash_attention: bool | None = None
88
+ threads: int | None = None
89
+ batch_size: int | None = None
90
+ keep_alive: str | None = None
91
+ provider_options: dict[str, Any] = Field(default_factory=dict)
92
+
93
+ @property
94
+ def profile_hash(self) -> str:
95
+ """Return this profile's 16-character hash, computed by the domain type that defines it.
96
+
97
+ Public because a consumer needs it for the same reason a producer does: two results are
98
+ comparable only if their profiles hash identically
99
+ (ADR-0023), and a reader holding a
100
+ payload should not have to reconstruct a :class:`baseaicore.RuntimeProfile` by hand to
101
+ find that out. Delegating rather than reimplementing is the point — a second
102
+ implementation of this hash would eventually disagree with the first, and the whole value
103
+ of the hash is that it does not.
104
+
105
+ Returns:
106
+ 16 lowercase hex characters, identical to what
107
+ :attr:`baseaicore.RuntimeProfile.profile_hash` computes for the same field values.
108
+ """
109
+ return RuntimeProfile(
110
+ context_size=self.context_size,
111
+ kv_cache_precision=self.kv_cache_precision,
112
+ gpu_layers=self.gpu_layers,
113
+ flash_attention=self.flash_attention,
114
+ threads=self.threads,
115
+ batch_size=self.batch_size,
116
+ keep_alive=self.keep_alive,
117
+ provider_options=dict(self.provider_options),
118
+ ).profile_hash
119
+
120
+
121
+ class ApplicationProvenanceFields(PayloadDefinition):
122
+ """Which build of the producing application measured this result.
123
+
124
+ All three fields are part of Machine Identity §6's minimum provenance set — none is optional
125
+ enrichment — because "which code produced this number" is exactly what a regression hunt needs
126
+ first.
127
+
128
+ Attributes:
129
+ name: The producing application's distribution name, e.g. ``"freeweight"``.
130
+ version: That application's version string.
131
+ git_commit: The commit the running build was checked out at.
132
+ """
133
+
134
+ name: str = Field(min_length=1)
135
+ version: str = Field(min_length=1)
136
+ git_commit: str = Field(min_length=1)
137
+
138
+
139
+ class PromptUsageFields(PayloadDefinition):
140
+ """One prompt this benchmark used, identified precisely enough to reproduce it.
141
+
142
+ Attributes:
143
+ prompt_id: The prompt's identifier in its manifest.
144
+ version: The prompt's own version.
145
+ sha256: The prompt's content hash, as computed by whatever produced it — not validated
146
+ against a fixed hex format here; see the module docstring's note on hash-shaped
147
+ fields.
148
+ """
149
+
150
+ prompt_id: str = Field(min_length=1)
151
+ version: str = Field(min_length=1)
152
+ sha256: str = Field(min_length=1)
153
+
154
+
155
+ class BenchmarkSuiteProvenanceFields(PayloadDefinition):
156
+ """Which benchmark suite produced this result, and at what version.
157
+
158
+ A differing ``suite_version``, ``dataset_hashes`` entry or ``prompt_subset_hash`` is a hard
159
+ separation for confidence purposes, never a discount
160
+ (ADR-0017) — SetSpec carries
161
+ these values; the separation itself is LoadCoach's and FreeWeight's behaviour to apply.
162
+
163
+ Attributes:
164
+ suite_key: The suite's identifier, e.g. ``"native.tool_use"``.
165
+ suite_version: The suite's own version. A version bump separates results from different
166
+ versions; they are never averaged together.
167
+ category: The suite's category, e.g. ``"tool_use"``.
168
+ runner: How the suite executes, e.g. ``"native"`` or ``"external"``.
169
+ manifest_hash: Hash of the suite's manifest, as the producer computed it.
170
+ dataset_hashes: Hash per named dataset the suite depends on; empty when the suite uses
171
+ none.
172
+ prompt_subset_hash: Hash of only the prompts *this suite* declares — the fingerprint
173
+ input, per benchmark and not per pack
174
+ (ADR-0028).
175
+ prompts_used: Every prompt named by ``prompt_subset_hash``, individually identified.
176
+ """
177
+
178
+ suite_key: str = Field(min_length=1)
179
+ suite_version: str = Field(min_length=1)
180
+ category: str | None = None
181
+ runner: str | None = None
182
+ manifest_hash: str = Field(min_length=1)
183
+ dataset_hashes: dict[str, str] = Field(default_factory=dict)
184
+ prompt_subset_hash: str = Field(min_length=1)
185
+ prompts_used: WireSequence[PromptUsageFields] = ()
186
+
187
+
188
+ class ServedContextSource(StrEnum):
189
+ """Where a run's actually-served context length came from.
190
+
191
+ Distinct from :attr:`~setspec.model.v1.ModelIdentityFields.max_context`, which is what the
192
+ model *advertises*, not what a provider was actually configured to serve
193
+ (ADR-0023 §4).
194
+ """
195
+
196
+ CONFIGURED = "configured"
197
+ """An operator or the runtime profile explicitly set it."""
198
+
199
+ REPORTED = "reported"
200
+ """The provider reported the context it actually served."""
201
+
202
+ ASSUMED = "assumed"
203
+ """Neither configured nor reported; taken from the model's advertised default."""
204
+
205
+
206
+ class ExecutionProvenanceFields(PayloadDefinition):
207
+ """The execution parameters actually in effect for this result, after precedence resolution.
208
+
209
+ Attributes:
210
+ effective_parameters: Resolved sampling and limit parameters, post-precedence-chain —
211
+ what was actually used, not what any one layer requested.
212
+ repetitions: How many repetitions this result's configuration requested.
213
+ sample_count: How many samples this result actually produced.
214
+ seed: The seed used, or the literal string ``"nondeterministic"`` when none was set.
215
+ served_context: The context length actually served.
216
+ served_context_source: Where that value came from.
217
+ gpu_index: The device this result's metrics are attributed to.
218
+ multi_gpu_visible: Whether more than one GPU was visible during measurement
219
+ (ADR-0027).
220
+ """
221
+
222
+ effective_parameters: dict[str, Any] = Field(default_factory=dict)
223
+ repetitions: int = Field(ge=1)
224
+ sample_count: int = Field(ge=0)
225
+ seed: int | str
226
+ served_context: int = Field(ge=0)
227
+ served_context_source: WireEnum[ServedContextSource]
228
+ gpu_index: int = Field(ge=0, default=0)
229
+ multi_gpu_visible: bool = False
230
+
231
+ @model_validator(mode="after")
232
+ def _check_seed(self) -> Self:
233
+ """Require a string seed to be exactly the documented sentinel.
234
+
235
+ Raises:
236
+ ValueError: If ``seed`` is a string other than ``"nondeterministic"`` — a numeric
237
+ string masquerading as a seed is a producer bug the schema should catch, not
238
+ silently coerce.
239
+ """
240
+ if isinstance(self.seed, str) and self.seed != "nondeterministic":
241
+ raise ValueError(
242
+ f"seed must be an integer or the literal string 'nondeterministic'; got "
243
+ f"{self.seed!r}. A run that did not set a seed says so with the sentinel, not "
244
+ "with any other string."
245
+ )
246
+ return self
247
+
248
+
249
+ class ReproducibilityFingerprintFields(PayloadDefinition):
250
+ """The answer to "could this measurement be repeated, and is that other result the same
251
+ thing?"
252
+ (Machine Identity §4).
253
+
254
+ Attributes:
255
+ reproducibility_fingerprint: The hash itself, as the producer computed it.
256
+ fingerprint_document: The **full input document that was hashed**, stored verbatim per
257
+ Machine Identity §4 rule 2 — "a hash you cannot explain is useless during a regression
258
+ hunt." Kept as a structured but untyped mapping rather than a fully modeled nested
259
+ structure: its shape mirrors provenance already typed elsewhere on this result, and a
260
+ second, independently-typed copy of that shape would drift from the first the moment
261
+ either one changed.
262
+ """
263
+
264
+ reproducibility_fingerprint: str = Field(min_length=1)
265
+ fingerprint_document: dict[str, Any]
266
+
267
+
268
+ class MeasurementClass(StrEnum):
269
+ """Whether the model was cold, warm, or serving a reused cache when this was measured.
270
+
271
+ The cold/warm marker
272
+ Machine Identity §6
273
+ calls "optional but strongly recommended". For a load-time or first-token metric it is not
274
+ optional in practice: the same benchmark against the same weights reports figures that differ
275
+ by an order of magnitude depending on this value alone, so a result that omits it is a number
276
+ two readers will interpret differently.
277
+ """
278
+
279
+ COLD = "cold"
280
+ """Measured on a fresh load, with nothing cached."""
281
+
282
+ WARM = "warm"
283
+ """Measured with the model already resident."""
284
+
285
+ CACHE_REUSED = "cache_reused"
286
+ """Measured while a prompt or KV cache from an earlier request was still in play."""
287
+
288
+ NOT_APPLICABLE = "n/a"
289
+ """The distinction does not apply to what this benchmark measures."""
290
+
291
+
292
+ class TelemetrySummaryFields(PayloadDefinition):
293
+ """What the machine was doing while this benchmark ran, reduced to the figures that explain it.
294
+
295
+ The telemetry summary
296
+ Machine Identity §6
297
+ names as "required for any benchmark whose numbers depend on them" — a memory or energy
298
+ benchmark's result is not interpretable without it, and a throttled run's timings mean
299
+ something different from an unthrottled one's.
300
+
301
+ Every figure is device-scoped by the enclosing result's ``execution.gpu_index``, never
302
+ machine-wide: there is no meaningful sum of two GPUs' VRAM or power, and
303
+ ADR-0027 requires each derived figure to name
304
+ the device it came from rather than silently aggregating across a machine the reference
305
+ hardware does not have.
306
+
307
+ Attributes:
308
+ peak_vram_bytes: Highest device memory in use during the benchmark.
309
+ peak_power_watts: Highest instantaneous draw observed.
310
+ mean_power_watts: Mean draw across the benchmark — the figure an energy estimate derives
311
+ from, and never interchangeable with the peak.
312
+ max_temperature_c: Highest device temperature observed.
313
+ throttled: Whether the device throttled at any point. ``None`` when the driver exposes no
314
+ throttle-reason mask to read, which is a different statement from ``False``: guessing
315
+ "not throttled" would attribute a slow result to the model rather than to the hardware
316
+ protecting itself.
317
+ """
318
+
319
+ peak_vram_bytes: MeasurementField = UNSUPPORTED
320
+ peak_power_watts: MeasurementField = UNSUPPORTED
321
+ mean_power_watts: MeasurementField = UNSUPPORTED
322
+ max_temperature_c: MeasurementField = UNSUPPORTED
323
+ throttled: bool | None = None
324
+
325
+
326
+ class BenchmarkResultStatus(StrEnum):
327
+ """Terminal states of one benchmark within a run (``run_tests.status`` in FreeWeight's data
328
+ model). A failed benchmark never fails its run; a failed sample never fails its benchmark."""
329
+
330
+ COMPLETED = "completed"
331
+ FAILED = "failed"
332
+ SKIPPED = "skipped"
333
+ CANCELLED = "cancelled"
334
+
335
+
336
+ class BenchmarkResultFields(PayloadDefinition):
337
+ """Field definitions for ``benchmark.result``; use :data:`BenchmarkResultOut` /
338
+ :data:`BenchmarkResultIn`.
339
+
340
+ One benchmark, one measurement subject, metrics plus provenance plus a samples reference
341
+ ([spec §7](../../../docs/packages/setspec/spec.md)). The provenance fields are exactly Machine
342
+ Identity §6's minimum set — see the module docstring for how each bullet there maps onto a
343
+ field or nested object here.
344
+
345
+ Attributes:
346
+ model: The measured weights, plus what the provider reports about them.
347
+ runtime_profile: How the provider was asked to serve the model for this measurement.
348
+ runtime_profile_hash: :attr:`runtime_profile`'s hash — checked, not merely carried; see
349
+ :meth:`_check_runtime_profile_hash`.
350
+ machine_fingerprint: Where this was measured.
351
+ machine_profile: A full snapshot of the machine at measurement time, when the producer
352
+ chose to embed one rather than carry only the fingerprint.
353
+ suite: Which benchmark suite produced this, and at what version.
354
+ execution: The execution parameters actually in effect.
355
+ environment: Provider and drift-sensitive facts at measurement time.
356
+ application: Which build of the producing application measured this.
357
+ reproducibility: The fingerprint proving this measurement could be repeated.
358
+ started_at: When measurement began.
359
+ completed_at: When measurement ended; never earlier than ``started_at``.
360
+ status: This benchmark's terminal state.
361
+ skip_reason: Present iff ``status`` is ``skipped``.
362
+ error_code: A stable, machine-readable reason this benchmark did not complete. Optional
363
+ because not every non-completion has one — a cancelled benchmark was simply stopped —
364
+ but a ``failed`` result that omits it is a failure nobody can triage.
365
+ error_text: Human-readable detail alongside ``error_code``.
366
+ completed_cases: How many of this benchmark's cases finished.
367
+ total_cases: How many cases this benchmark set out to run. Never smaller than
368
+ ``completed_cases`` — see :meth:`_check_case_counts`.
369
+ measurement_class: Whether this was measured cold, warm, or against a reused cache.
370
+ telemetry_summary: What the hardware was doing while this ran.
371
+ raw_response_ref: An opaque, producer-local pointer to the stored raw provider response.
372
+ metrics: Every metric this benchmark measured. Empty for a skipped or cancelled result;
373
+ a completed result with none is rejected — see :meth:`_check_status_coherence`.
374
+ samples_ref: An opaque, producer-local pointer to the raw sample rows behind ``metrics``.
375
+ Never resolved by a consumer; carried only for the producer's own drill-down.
376
+
377
+ The last six are the provenance
378
+ Machine Identity §6
379
+ calls "optional but strongly recommended, and required for any benchmark whose numbers depend
380
+ on them", plus the two error fields FreeWeight's own ``run_tests`` row carries. They are
381
+ declared rather than left out because the writer half of this pair forbids unknown keys: a
382
+ field this schema does not name is a field a producer *cannot emit at all*, so omitting an
383
+ optional field is not a neutral act here — it is a decision that the field may never be sent.
384
+ """
385
+
386
+ model: ModelIdentityFields
387
+ runtime_profile: RuntimeProfileFields
388
+ runtime_profile_hash: str = Field(min_length=1)
389
+ machine_fingerprint: str = Field(min_length=1)
390
+ machine_profile: MachineProfileFields | None = None
391
+ suite: BenchmarkSuiteProvenanceFields
392
+ execution: ExecutionProvenanceFields
393
+ environment: EnvironmentFields
394
+ application: ApplicationProvenanceFields
395
+ reproducibility: ReproducibilityFingerprintFields
396
+ started_at: TimestampField
397
+ completed_at: TimestampField
398
+ status: WireEnum[BenchmarkResultStatus]
399
+ skip_reason: str | None = None
400
+ error_code: str | None = None
401
+ error_text: str | None = None
402
+ completed_cases: int | None = Field(default=None, ge=0)
403
+ total_cases: int | None = Field(default=None, ge=0)
404
+ measurement_class: WireEnum[MeasurementClass] | None = None
405
+ telemetry_summary: TelemetrySummaryFields | None = None
406
+ raw_response_ref: str | None = None
407
+ metrics: WireSequence[MetricValueFields] = ()
408
+ samples_ref: str | None = None
409
+
410
+ @model_validator(mode="after")
411
+ def _check_runtime_profile_hash(self) -> Self:
412
+ """Recompute ``runtime_profile_hash`` from the embedded profile and require agreement.
413
+
414
+ Safe to recompute for the same reason ``canonical_id`` is: a runtime profile's hash is a
415
+ pure function of its own fields with no historical-policy caveat, unlike a machine
416
+ fingerprint.
417
+
418
+ Raises:
419
+ ValueError: If the declared hash does not match what ``runtime_profile`` recomputes.
420
+ """
421
+ recomputed = self.runtime_profile.profile_hash
422
+ if recomputed != self.runtime_profile_hash:
423
+ raise ValueError(
424
+ f"runtime_profile_hash {self.runtime_profile_hash!r} does not match "
425
+ f"runtime_profile, which recomputes to {recomputed!r}. The hash is a pure "
426
+ "function of the profile's own fields (ADR-0023) and is carried on the wire for "
427
+ "convenience, not as an independent fact."
428
+ )
429
+ return self
430
+
431
+ @model_validator(mode="after")
432
+ def _check_timing_order(self) -> Self:
433
+ """Require ``completed_at`` not to precede ``started_at``.
434
+
435
+ Raises:
436
+ ValueError: If the two timestamps are out of order.
437
+ """
438
+ if self.completed_at < self.started_at:
439
+ raise ValueError(
440
+ f"completed_at ({self.completed_at.isoformat()}) precedes started_at "
441
+ f"({self.started_at.isoformat()}); a benchmark cannot finish before it started."
442
+ )
443
+ return self
444
+
445
+ @model_validator(mode="after")
446
+ def _check_case_counts(self) -> Self:
447
+ """Require ``completed_cases`` not to exceed ``total_cases`` when both are present.
448
+
449
+ The same arithmetic honesty :meth:`_check_timing_order` applies to timestamps: "12 of 10
450
+ cases done" is not a progress report, it is a producer bug, and a consumer rendering it as
451
+ a percentage would show something above 100%.
452
+
453
+ Raises:
454
+ ValueError: If both counts are present and ``completed_cases`` exceeds
455
+ ``total_cases``.
456
+ """
457
+ if (
458
+ self.completed_cases is not None
459
+ and self.total_cases is not None
460
+ and self.completed_cases > self.total_cases
461
+ ):
462
+ raise ValueError(
463
+ f"completed_cases ({self.completed_cases}) exceeds total_cases "
464
+ f"({self.total_cases}); a benchmark cannot complete more cases than it ran."
465
+ )
466
+ return self
467
+
468
+ @model_validator(mode="after")
469
+ def _check_status_coherence(self) -> Self:
470
+ """Require ``skip_reason`` and ``metrics`` to agree with ``status``.
471
+
472
+ Raises:
473
+ ValueError: If ``skip_reason`` is set without ``status == "skipped"``, if a skipped
474
+ result carries no ``skip_reason``, or if a completed result reports no metrics at
475
+ all — a completed benchmark that measured nothing is not a completed benchmark.
476
+ """
477
+ if self.status is BenchmarkResultStatus.SKIPPED and self.skip_reason is None:
478
+ raise ValueError("a skipped result must name its skip_reason")
479
+ if self.status is not BenchmarkResultStatus.SKIPPED and self.skip_reason is not None:
480
+ raise ValueError(
481
+ f"skip_reason is set to {self.skip_reason!r} but status is {self.status.value!r}, "
482
+ "not 'skipped'; skip_reason is only meaningful for a skipped result"
483
+ )
484
+ if self.status is BenchmarkResultStatus.COMPLETED and not self.metrics:
485
+ raise ValueError(
486
+ "status is 'completed' but metrics is empty; a completed benchmark that measured "
487
+ "nothing is not a completed benchmark"
488
+ )
489
+ return self
490
+
491
+
492
+ BenchmarkResultOut, BenchmarkResultIn = payload_models(BenchmarkResultFields)
493
+ """The ``benchmark.result`` payload pair: ``Out`` for writers, ``In`` for readers."""
494
+
495
+
496
+ class RunStatus(StrEnum):
497
+ """A run's state, mirroring FreeWeight's run state machine (data model §3).
498
+
499
+ ``INTERRUPTED`` is distinct from ``FAILED``: it means the process died, is discovered at
500
+ startup recovery, and is resumable with completed benchmarks preserved — a run summary
501
+ carrying it is not reporting a failure, it is reporting an unfinished run.
502
+ """
503
+
504
+ QUEUED = "queued"
505
+ PREPARING = "preparing"
506
+ WARMING = "warming"
507
+ RUNNING = "running"
508
+ COMPLETED = "completed"
509
+ FAILED = "failed"
510
+ CANCELLING = "cancelling"
511
+ CANCELLED = "cancelled"
512
+ INTERRUPTED = "interrupted"
513
+
514
+
515
+ class BenchmarkRunSummaryFields(PayloadDefinition):
516
+ """Field definitions for ``benchmark.run_summary``; use :data:`BenchmarkRunSummaryOut` /
517
+ :data:`BenchmarkRunSummaryIn`.
518
+
519
+ One run: subject, suite, status, timings, aggregate metrics
520
+ ([spec §7](../../../docs/packages/setspec/spec.md)) — deliberately lighter than
521
+ :class:`BenchmarkResultFields`, which carries one benchmark's full per-result provenance; a run
522
+ summary is the roll-up of many such results and reuses the same subject/suite/environment
523
+ building blocks rather than repeating a full provenance set per aggregate metric.
524
+
525
+ Attributes:
526
+ model: The measurement subject's model identity.
527
+ runtime_profile: The runtime profile every benchmark in this run measured under.
528
+ runtime_profile_hash: Checked against ``runtime_profile``, as on a result.
529
+ machine_fingerprint: Where this run executed.
530
+ suite: The suite this run executed.
531
+ environment: Provider and drift-sensitive facts for the run.
532
+ application: Which build of the producing application ran this.
533
+ reproducibility: The run-level reproducibility fingerprint.
534
+ status: The run's current or terminal state.
535
+ created_at: When the run was created (queued).
536
+ started_at: When execution began; ``None`` if the run never left ``queued``.
537
+ completed_at: When execution ended; ``None`` while still in progress.
538
+ aggregate_metrics: Run-level rolled-up metrics, distinct from any one benchmark's own.
539
+ error_code: Present iff the run ended in a state a caller should investigate.
540
+ error_text: Human-readable detail alongside ``error_code``.
541
+ """
542
+
543
+ model: ModelIdentityFields
544
+ runtime_profile: RuntimeProfileFields
545
+ runtime_profile_hash: str = Field(min_length=1)
546
+ machine_fingerprint: str = Field(min_length=1)
547
+ suite: BenchmarkSuiteProvenanceFields
548
+ environment: EnvironmentFields
549
+ application: ApplicationProvenanceFields
550
+ reproducibility: ReproducibilityFingerprintFields
551
+ status: WireEnum[RunStatus]
552
+ created_at: TimestampField
553
+ started_at: TimestampField | None = None
554
+ completed_at: TimestampField | None = None
555
+ aggregate_metrics: WireSequence[MetricValueFields] = ()
556
+ error_code: str | None = None
557
+ error_text: str | None = None
558
+
559
+ @model_validator(mode="after")
560
+ def _check_runtime_profile_hash(self) -> Self:
561
+ """Recompute ``runtime_profile_hash`` from the embedded profile and require agreement.
562
+
563
+ Raises:
564
+ ValueError: If the declared hash does not match what ``runtime_profile`` recomputes.
565
+ """
566
+ recomputed = self.runtime_profile.profile_hash
567
+ if recomputed != self.runtime_profile_hash:
568
+ raise ValueError(
569
+ f"runtime_profile_hash {self.runtime_profile_hash!r} does not match "
570
+ f"runtime_profile, which recomputes to {recomputed!r}."
571
+ )
572
+ return self
573
+
574
+ @model_validator(mode="after")
575
+ def _check_timing_order(self) -> Self:
576
+ """Require ``completed_at`` not to precede ``started_at`` when both are present.
577
+
578
+ Raises:
579
+ ValueError: If both timestamps are present and out of order.
580
+ """
581
+ if (
582
+ self.started_at is not None
583
+ and self.completed_at is not None
584
+ and self.completed_at < self.started_at
585
+ ):
586
+ raise ValueError(
587
+ f"completed_at ({self.completed_at.isoformat()}) precedes started_at "
588
+ f"({self.started_at.isoformat()})."
589
+ )
590
+ return self
591
+
592
+
593
+ BenchmarkRunSummaryOut, BenchmarkRunSummaryIn = payload_models(BenchmarkRunSummaryFields)
594
+ """The ``benchmark.run_summary`` payload pair: ``Out`` for writers, ``In`` for readers."""