setspec 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
setspec/goal/v1.py ADDED
@@ -0,0 +1,390 @@
1
+ """Contract module — ``benchmark.goal_pack`` and ``benchmark.calibration_report`` v1.
2
+
3
+ Imports pydantic and :mod:`baseaicore`; performs no I/O. These two payloads carry FreeWeight's
4
+ user-authored goal benchmarks across a boundary: a goal pack so a rubric can move between machines
5
+ and be re-run verbatim, and a calibration report so the judge's measured agreement with its author
6
+ travels with — and can be audited apart from — the scores it produced
7
+ (ADR-0031, ADR-0032).
8
+
9
+ **The rule these schemas exist to make transportable.** A judged score is a measurement only when
10
+ the instrument that produced it has been characterized against ground truth. Everything in
11
+ ``benchmark.calibration_report`` describes the *judge's* error, never the measured model's
12
+ performance, and ``kappa_w`` is never carried without ``n_holdout`` — a coefficient without its
13
+ sample count is a number pretending to be a fact.
14
+
15
+ **Status: draft (`1.0`).** Registered in :data:`setspec.envelope.SUPPORTED_SCHEMAS` and listed in
16
+ :data:`setspec.envelope.DRAFT_SCHEMAS`: these shapes predate FreeWeight Phases 8A–8B actually
17
+ producing goal packs and calibration reports, so the freeze against real output may still adjust a
18
+ field that turns out to be shaped wrong.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ from enum import StrEnum
24
+ from typing import Self
25
+
26
+ from pydantic import Field, model_validator
27
+
28
+ from setspec.base import PayloadDefinition, WireEnum, WireSequence, payload_models
29
+ from setspec.capability.v1 import CalibrationFields, JudgeSetFields
30
+ from setspec.serialization import TimestampField
31
+
32
+ __all__ = [
33
+ "CalibrationReportFields",
34
+ "CalibrationReportIn",
35
+ "CalibrationReportOut",
36
+ "CriterionAgreementFields",
37
+ "GoalCriterionFields",
38
+ "GoalPackFields",
39
+ "GoalPackIn",
40
+ "GoalPackOut",
41
+ "GoalTaskFields",
42
+ "ScoringRung",
43
+ ]
44
+
45
+ _MINIMUM_WEIGHT = 0.0
46
+ _WEIGHT_SUM_TOLERANCE = 1e-6
47
+ _MINIMUM_SCALE_POINTS = 3
48
+ _MAXIMUM_SCALE_POINTS = 7
49
+
50
+
51
+ class ScoringRung(StrEnum):
52
+ """Which rung of the scoring ladder a criterion is scored at (benchmark catalog §1).
53
+
54
+ A ``StrEnum`` so it serializes as its own name. Recording it per criterion is not bookkeeping:
55
+ it is what makes ``score_method_mix`` computable, and therefore what lets a consumer tell a
56
+ score that is mostly rules from a score that is mostly judgement. Those are different kinds of
57
+ number and presenting them identically is the failure this field prevents.
58
+ """
59
+
60
+ RULE = "rule"
61
+ """Deterministic check over the output text alone. Free, exact, and never disagrees with you."""
62
+
63
+ REFERENCE = "reference"
64
+ """Deterministic check against user-supplied ground truth: an annotated source or claim list."""
65
+
66
+ HUMAN = "human"
67
+ """The author graded it themselves, blinded. Validity is 1.0 by definition."""
68
+
69
+ JUDGE = "judge"
70
+ """A jury of models graded it. The only rung that requires calibration to mean anything."""
71
+
72
+
73
+ class GoalCriterionFields(PayloadDefinition):
74
+ """One measurable quality within a goal, and how it is scored.
75
+
76
+ Attributes:
77
+ key: Stable identifier within the goal. Never renamed — a rename is a new criterion,
78
+ because a renamed criterion whose history merged with the old one would silently
79
+ compare two different measurements.
80
+ name: Human-readable label. Display only, and deliberately **not** a ``goal_hash`` input:
81
+ renaming a criterion for readability must not separate a year of results, while
82
+ changing what it checks must.
83
+ rung: Which ladder rung scores it.
84
+ weight: Its share of the composite, positive and summing to ``1`` across the goal.
85
+ is_gate: Whether failing it zeroes the sample's composite outright. For disqualifying
86
+ properties — a forbidden phrase, invalid JSON — rather than gradual ones.
87
+ rule_type: For a ``rule`` or ``reference`` criterion, which check runs, e.g.
88
+ ``"forbidden_phrases"``. ``None`` for ``human`` and ``judge``.
89
+ scale_points: For a ``judge`` or ``human`` criterion, the ordinal scale's size — 3, 5 or
90
+ 7. ``None`` for a deterministic criterion.
91
+ has_scale_descriptors: Whether the ordinal scale carries anchoring descriptors. Always
92
+ ``True`` on a valid judged criterion: an unanchored scale ("rate the tone 1-5")
93
+ reliably produces agreement near zero, so FreeWeight refuses one at authoring time.
94
+ Carried on the wire so an importer can see the rubric was anchored without holding
95
+ the descriptors themselves.
96
+ """
97
+
98
+ key: str = Field(min_length=1)
99
+ name: str = Field(min_length=1)
100
+ rung: WireEnum[ScoringRung]
101
+ weight: float = Field(gt=_MINIMUM_WEIGHT, le=1.0)
102
+ is_gate: bool = False
103
+ rule_type: str | None = Field(default=None, min_length=1)
104
+ scale_points: int | None = Field(
105
+ default=None, ge=_MINIMUM_SCALE_POINTS, le=_MAXIMUM_SCALE_POINTS
106
+ )
107
+ has_scale_descriptors: bool = False
108
+
109
+ @model_validator(mode="after")
110
+ def _check_rung_shape(self) -> Self:
111
+ """Require the fields a criterion's rung actually needs, and refuse the ones it cannot use.
112
+
113
+ Raises:
114
+ ValueError: If a deterministic criterion names no ``rule_type``; if a judged or human
115
+ criterion declares no ``scale_points``; if a judged criterion has no scale
116
+ descriptors; or if a criterion carries a field belonging to a different rung.
117
+ """
118
+ deterministic = self.rung in (ScoringRung.RULE, ScoringRung.REFERENCE)
119
+ graded = self.rung in (ScoringRung.JUDGE, ScoringRung.HUMAN)
120
+ if deterministic and self.rule_type is None:
121
+ raise ValueError(
122
+ f"criterion {self.key!r} is scored at rung {self.rung.value!r} but names no "
123
+ "rule_type. A deterministic criterion is defined by the check it runs."
124
+ )
125
+ if deterministic and self.scale_points is not None:
126
+ raise ValueError(
127
+ f"criterion {self.key!r} is deterministic but declares scale_points. An ordinal "
128
+ "scale belongs to a graded criterion; a rule returns a fraction."
129
+ )
130
+ if graded and self.scale_points is None:
131
+ raise ValueError(
132
+ f"criterion {self.key!r} is graded at rung {self.rung.value!r} but declares no "
133
+ "scale_points. A grade with no scale cannot be compared with another grader's."
134
+ )
135
+ if graded and self.rule_type is not None:
136
+ raise ValueError(
137
+ f"criterion {self.key!r} is graded but names a rule_type. If a rule can check it, "
138
+ "it belongs at rung 'rule' — that is what the authoring lint says, and encoding "
139
+ "both here would make the ladder position ambiguous."
140
+ )
141
+ if self.rung is ScoringRung.JUDGE and not self.has_scale_descriptors:
142
+ raise ValueError(
143
+ f"criterion {self.key!r} is judged but its scale has no descriptors. An "
144
+ "unanchored ordinal scale gives a jury nothing to calibrate against and reliably "
145
+ "produces agreement near zero, so it is refused at authoring time rather than "
146
+ "discovered after the author has graded twelve samples (ADR-0031 §3)."
147
+ )
148
+ return self
149
+
150
+
151
+ class GoalTaskFields(PayloadDefinition):
152
+ """One task the candidate model answers, identified by the prompt that produced it.
153
+
154
+ Attributes:
155
+ key: Stable identifier within the goal.
156
+ prompt_id: The task prompt's record ID (ADR-0012 — a task prompt is a prompt record).
157
+ prompt_version: That record's semantic version.
158
+ prompt_sha256: That record's canonical hash, a ``goal_hash`` input.
159
+ is_starter: Whether this is unedited shipped starter content. A goal still running
160
+ entirely on starter tasks measures the starter's author's work rather than the
161
+ importer's, and carrying the flag is what lets a consumer say so.
162
+ """
163
+
164
+ key: str = Field(min_length=1)
165
+ prompt_id: str = Field(min_length=1)
166
+ prompt_version: str = Field(min_length=1)
167
+ prompt_sha256: str = Field(min_length=1)
168
+ is_starter: bool = False
169
+
170
+
171
+ class GoalPackFields(PayloadDefinition):
172
+ """Field definitions for ``benchmark.goal_pack``; use :data:`GoalPackOut` / :data:`GoalPackIn`.
173
+
174
+ The portable form of a user-authored goal (ADR-0031 §6): enough to re-run the same measurement
175
+ on another machine, and enough for a consumer to decide comparability without asking the
176
+ producer.
177
+
178
+ The author's calibration *grades* are deliberately **not** here. They are the subjective
179
+ ground truth and they are large; what travels is the goal's definition plus the agreement it
180
+ achieved, in :class:`CalibrationReportFields`. An importer who wants the rubric held to their
181
+ own taste re-calibrates against their own grades, which is the more honest default anyway.
182
+
183
+ Attributes:
184
+ slug: The goal's stable identifier. Its capability is ``user.<slug>``.
185
+ name: Human-readable label.
186
+ intent: The author's own description of what they were trying to get. Not machine-read
187
+ and not a ``goal_hash`` input — it exists so the goal is legible in six months.
188
+ goal_pack_version: Semantic version of this pack. A **major** bump for any change inside
189
+ ``goal_hash``.
190
+ goal_hash: The measurement-defining hash: criteria, weights, rungs, rule parameters, scale
191
+ descriptors, task prompt hashes, the judge prompt hash and the jury configuration.
192
+ Excludes display names, ``intent`` and ``contributes_to``.
193
+ contributes_to: An existing capability root this goal also feeds, or ``None``. When set,
194
+ FreeWeight emits evidence **twice** — once as ``user.<slug>`` keeping the goal's
195
+ identity, once as a weighted source inside the shipped capability — and never only as
196
+ the shipped one, which would fold one person's taste into a term other components
197
+ believe is objective (ADR-0032 §1).
198
+ criteria: The goal's criteria. At least one, weights summing to ``1``.
199
+ tasks: The tasks candidates answer. At least one.
200
+ judge_set: The jury configuration, when any criterion is judged.
201
+ unforked: Whether the criteria and tasks are unedited starter content.
202
+ created_by: Free text the author supplied. Never harvested from the environment.
203
+ created_at: When the pack was authored.
204
+ """
205
+
206
+ slug: str = Field(min_length=1, pattern=r"^[a-z][a-z0-9_]*$")
207
+ name: str = Field(min_length=1)
208
+ intent: str = ""
209
+ goal_pack_version: str = Field(min_length=1)
210
+ goal_hash: str = Field(min_length=1)
211
+ contributes_to: str | None = Field(default=None, min_length=1)
212
+ criteria: WireSequence[GoalCriterionFields] = ()
213
+ tasks: WireSequence[GoalTaskFields] = ()
214
+ judge_set: JudgeSetFields | None = None
215
+ unforked: bool = False
216
+ created_by: str = Field(min_length=1)
217
+ created_at: TimestampField
218
+
219
+ @model_validator(mode="after")
220
+ def _check_pack_is_runnable(self) -> Self:
221
+ """Require a pack to define a measurement that could actually be taken.
222
+
223
+ Raises:
224
+ ValueError: If there are no criteria or no tasks, if criterion keys collide, if the
225
+ weights do not sum to ``1``, or if a judged criterion exists with no jury to
226
+ score it.
227
+ """
228
+ if not self.criteria:
229
+ raise ValueError(
230
+ f"goal {self.slug!r} declares no criteria. A goal with nothing to measure is not "
231
+ "a benchmark."
232
+ )
233
+ if not self.tasks:
234
+ raise ValueError(
235
+ f"goal {self.slug!r} declares no tasks. Criteria score outputs; with no task "
236
+ "there is no output to score."
237
+ )
238
+ keys = [criterion.key for criterion in self.criteria]
239
+ duplicates = sorted({key for key in keys if keys.count(key) > 1})
240
+ if duplicates:
241
+ raise ValueError(
242
+ f"goal {self.slug!r} declares criteria {duplicates} more than once. A criterion "
243
+ "key identifies a measurement over time, so a collision merges two of them."
244
+ )
245
+ total = sum(criterion.weight for criterion in self.criteria)
246
+ if abs(total - 1.0) > _WEIGHT_SUM_TOLERANCE:
247
+ raise ValueError(
248
+ f"goal {self.slug!r} has criterion weights summing to {total}, not 1. The "
249
+ "composite is a weighted mean, so weight that is unaccounted for changes every "
250
+ "criterion's real share rather than only the total."
251
+ )
252
+ judged = [c.key for c in self.criteria if c.rung is ScoringRung.JUDGE]
253
+ if judged and self.judge_set is None:
254
+ raise ValueError(
255
+ f"goal {self.slug!r} has judged criteria {judged} but no judge_set. A judged "
256
+ "score is a property of the jury that produced it; without the jury's identity "
257
+ "the result cannot be separated from one produced by a different instrument "
258
+ "(ADR-0032 §4)."
259
+ )
260
+ return self
261
+
262
+
263
+ GoalPackOut, GoalPackIn = payload_models(GoalPackFields)
264
+ """The ``benchmark.goal_pack`` payload pair: ``Out`` for writers, ``In`` for readers."""
265
+
266
+
267
+ class CriterionAgreementFields(PayloadDefinition):
268
+ """One judged criterion's measured agreement between the jury and the goal's author.
269
+
270
+ Attributes:
271
+ criterion_key: Which criterion this describes.
272
+ weight: Its share of the composite, so a reader can weight the agreement figures the same
273
+ way the score weighted the criteria.
274
+ agreement: The agreement statistics for this criterion, including the ``n_holdout`` that
275
+ makes ``kappa_w`` interpretable.
276
+ inter_juror_alpha: Krippendorff's alpha across jurors on this criterion, or ``None`` for a
277
+ single-juror jury — where the quantity does not exist rather than being zero. This is
278
+ what distinguishes jury bias from jury noise: high alpha with low ``kappa_w`` means
279
+ the jurors agree with each other and not with the author.
280
+ judge_validity_factor: This criterion's contribution to the goal's validity factor:
281
+ ``max(0, kappa_w) * min(1, sqrt(n_holdout / n_holdout_target))`` (ADR-0032 §2).
282
+ """
283
+
284
+ criterion_key: str = Field(min_length=1)
285
+ weight: float = Field(gt=_MINIMUM_WEIGHT, le=1.0)
286
+ agreement: CalibrationFields
287
+ inter_juror_alpha: float | None = Field(default=None, ge=-1.0, le=1.0)
288
+ judge_validity_factor: float = Field(ge=0.0, le=1.0)
289
+
290
+
291
+ class CalibrationReportFields(PayloadDefinition):
292
+ """Field definitions for ``benchmark.calibration_report``; use :data:`CalibrationReportOut` /
293
+ :data:`CalibrationReportIn`.
294
+
295
+ How well the jury agreed with the person whose goal this is, per criterion and weighted, and
296
+ whether that was enough to let the goal emit capability evidence at all.
297
+
298
+ This payload exists separately from ``capability.evidence`` because it is meaningful when no
299
+ evidence was emitted. A failed gate is the case a user most needs to see and the case the
300
+ evidence contract deliberately says nothing about: below the threshold FreeWeight emits no
301
+ evidence record, so without this report the most informative outcome would be invisible on the
302
+ wire (ADR-0032 §3).
303
+
304
+ Attributes:
305
+ goal_slug: Which goal was calibrated.
306
+ goal_hash: The exact rubric the agreement was measured against. Agreement measured on one
307
+ rubric says nothing about another, so this is not optional provenance.
308
+ judge_set: The jury the agreement was measured for. Change the jury and the report no
309
+ longer applies.
310
+ criteria: Per-criterion agreement, for judged criteria only. Empty when a goal is scored
311
+ entirely by rules — a legitimate and desirable state, not a failure.
312
+ weighted_kappa_w: Weighted across judged criteria; the value the gate compares. ``None``
313
+ when there are no judged criteria to weight.
314
+ min_agreement: The gate threshold in force, recorded because it is configuration and a
315
+ reader must not assume the default.
316
+ passed_gate: Whether evidence was emitted. ``True`` for a goal with no judged criteria:
317
+ nothing needed calibrating, so nothing failed to calibrate.
318
+ judge_validity_factor: The goal-level factor that multiplied into confidence.
319
+ n_anchor: Graded samples used as judge-prompt exemplars.
320
+ n_holdout: Graded samples withheld from the jury — the basis of every figure above.
321
+ partition_seed: The seed that produced the anchor/holdout split, so the partition is
322
+ reproducible and a reader can verify the holdout was not chosen to flatter the result.
323
+ graded_by: Free text identifying the grader.
324
+ measured_at: When this calibration was measured. Ages like evidence.
325
+ policy_version: The calibration-policy version these figures were computed under.
326
+ """
327
+
328
+ goal_slug: str = Field(min_length=1)
329
+ goal_hash: str = Field(min_length=1)
330
+ judge_set: JudgeSetFields | None = None
331
+ criteria: WireSequence[CriterionAgreementFields] = ()
332
+ weighted_kappa_w: float | None = Field(default=None, ge=-1.0, le=1.0)
333
+ min_agreement: float = Field(ge=-1.0, le=1.0)
334
+ passed_gate: bool
335
+ judge_validity_factor: float = Field(ge=0.0, le=1.0)
336
+ n_anchor: int = Field(ge=0)
337
+ n_holdout: int = Field(ge=0)
338
+ partition_seed: int
339
+ graded_by: str = Field(min_length=1)
340
+ measured_at: TimestampField
341
+ policy_version: str = Field(min_length=1)
342
+
343
+ @model_validator(mode="after")
344
+ def _check_report_coheres(self) -> Self:
345
+ """Require the report's verdict to follow from the figures it carries.
346
+
347
+ Raises:
348
+ ValueError: If a judged goal reports no weighted agreement; if the gate verdict
349
+ contradicts the threshold comparison; if a judged report names no jury; or if a
350
+ goal with judged criteria reports no holdout to have measured them on.
351
+ """
352
+ if self.criteria and self.weighted_kappa_w is None:
353
+ raise ValueError(
354
+ f"goal {self.goal_slug!r} reports per-criterion agreement but no "
355
+ "weighted_kappa_w. The weighted figure is what the gate compares, so a report "
356
+ "without it cannot explain its own verdict."
357
+ )
358
+ if self.criteria and self.judge_set is None:
359
+ raise ValueError(
360
+ f"goal {self.goal_slug!r} reports judged-criterion agreement with no judge_set. "
361
+ "Agreement is a property of a particular jury (ADR-0032 §4)."
362
+ )
363
+ if self.criteria and self.n_holdout < 1:
364
+ raise ValueError(
365
+ f"goal {self.goal_slug!r} reports judged-criterion agreement over "
366
+ f"{self.n_holdout} held-out samples. Agreement measured over nothing is not a "
367
+ "measurement, and a coefficient without a holdout to stand on is the exact "
368
+ "failure n_holdout travels everywhere to prevent."
369
+ )
370
+ if self.weighted_kappa_w is not None:
371
+ expected = self.weighted_kappa_w >= self.min_agreement
372
+ if self.passed_gate is not expected:
373
+ raise ValueError(
374
+ f"goal {self.goal_slug!r} reports passed_gate={self.passed_gate} with "
375
+ f"weighted_kappa_w={self.weighted_kappa_w} against min_agreement="
376
+ f"{self.min_agreement}. The verdict is the comparison; a report whose verdict "
377
+ "disagrees with its own numbers cannot be audited by the person it is for."
378
+ )
379
+ elif not self.passed_gate:
380
+ raise ValueError(
381
+ f"goal {self.goal_slug!r} has no judged criteria but reports passed_gate=False. "
382
+ "A goal scored entirely by rules has nothing to calibrate, so there is nothing "
383
+ "for it to fail — and marking it failed would penalize the most deterministic "
384
+ "rubric a user can write."
385
+ )
386
+ return self
387
+
388
+
389
+ CalibrationReportOut, CalibrationReportIn = payload_models(CalibrationReportFields)
390
+ """The ``benchmark.calibration_report`` payload pair: ``Out`` for writers, ``In`` for readers."""
File without changes
setspec/machine/v1.py ADDED
@@ -0,0 +1,131 @@
1
+ """Contract module — ``machine.profile`` v1: the static identity of the machine a run measured on.
2
+
3
+ Imports pydantic and :mod:`baseaicore`; performs no I/O. Exchange form of
4
+ :class:`baseaicore.MachineProfile` — a field-for-field mirror, including which fields are
5
+ required. A dataclass field with no Python default is required on the wire too; a field with a
6
+ default keeps that same default here, so a machine that could not report its core count produces
7
+ the identical payload whether it went through this schema or was read straight from BaseAiCore.
8
+
9
+ **Status: draft (`1.0`).** See :mod:`setspec.model.v1` for what that means and why.
10
+
11
+ **Deliberately not re-verified:** ``machine_fingerprint``. This is the one place a hash-shaped
12
+ field is carried without being recomputed and checked, and the asymmetry with
13
+ :mod:`setspec.model.v1` — which does recompute ``canonical_id`` — is the domain type's own
14
+ choice, not an oversight here. :class:`baseaicore.MachineProfile` documents its fingerprint as
15
+ "the *recorded* fingerprint, not a derived property... neither computed nor re-verified", because
16
+ the policy deciding which fields feed the fingerprint may change after a profile was written,
17
+ while a profile read back years later must still reconstruct exactly as stored. A ``canonical_id``
18
+ has no such caveat: it is a pure function of the identity triple under a format ADR-0024 fixes.
19
+ Recomputing a fingerprint here would therefore reject a historically valid profile the day that
20
+ policy changes, which is why this schema follows the domain type rather than its own sibling.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ from baseaicore import UNSUPPORTED, GpuVendor
26
+ from pydantic import Field
27
+
28
+ from setspec.base import PayloadDefinition, WireEnum, WireSequence, payload_models
29
+ from setspec.serialization import MeasurementField, TimestampField
30
+
31
+ __all__ = [
32
+ "GpuProfileFields",
33
+ "MachineProfileFields",
34
+ "MachineProfileIn",
35
+ "MachineProfileOut",
36
+ "StorageDeviceFields",
37
+ ]
38
+
39
+
40
+ class GpuProfileFields(PayloadDefinition):
41
+ """Static identity of one GPU, as a collector reported it.
42
+
43
+ Nested only — ``gpu.profile`` is not independently enveloped
44
+ (ADR-0009 lists no such schema), so this
45
+ is a plain :class:`~setspec.base.PayloadDefinition` rather than a generated pair; see that
46
+ class's docstring for why an embedded definition is always preserving.
47
+
48
+ Attributes:
49
+ index: The device's 0-based enumeration position; also the value a benchmark result
50
+ attributes a per-device measurement to
51
+ (ADR-0027).
52
+ name: The marketing name, e.g. ``"NVIDIA GeForce RTX 5060 Ti"``.
53
+ uuid: The device's stable hardware identifier, unchanged across reboots. ``None`` when the
54
+ collector could not read one.
55
+ vram_total_bytes: Total device memory — never "used", which is telemetry.
56
+ driver_version: The installed driver version. A drift signal, not identity.
57
+ cuda_version: The CUDA/ROCm toolkit version. A drift signal, not identity.
58
+ compute_capability: The device's compute capability, e.g. ``"12.0"``.
59
+ vendor: Who makes the device; ``UNKNOWN`` is the honest default, not a guess.
60
+ """
61
+
62
+ index: int = Field(ge=0)
63
+ name: str | None = None
64
+ uuid: str | None = None
65
+ vram_total_bytes: MeasurementField = UNSUPPORTED
66
+ driver_version: str | None = None
67
+ cuda_version: str | None = None
68
+ compute_capability: str | None = None
69
+ vendor: WireEnum[GpuVendor] = GpuVendor.UNKNOWN
70
+
71
+
72
+ class StorageDeviceFields(PayloadDefinition):
73
+ """A storage device attached to the machine — provenance only, excluded from the fingerprint.
74
+
75
+ Attributes:
76
+ name: The device name as the OS exposes it, e.g. ``"nvme0n1"``.
77
+ size_bytes: Total capacity.
78
+ model: The device's model string, if the OS exposes one.
79
+ rotational: ``True`` for spinning, ``False`` for solid state, ``None`` when the collector
80
+ could not tell — never guessed as ``False``.
81
+ """
82
+
83
+ name: str = Field(min_length=1)
84
+ size_bytes: MeasurementField = UNSUPPORTED
85
+ model: str | None = None
86
+ rotational: bool | None = None
87
+
88
+
89
+ class MachineProfileFields(PayloadDefinition):
90
+ """Field definitions for ``machine.profile``; use :data:`MachineProfileOut` /
91
+ :data:`MachineProfileIn`.
92
+
93
+ Attributes:
94
+ machine_fingerprint: The identity this profile was stored under.
95
+ hostname: The machine's hostname. Required as a key (may be ``null``): a producer must say
96
+ explicitly that it could not read one, not merely omit it — the schema-level half of
97
+ "provenance completeness enforced by the schema" (Phase 2 gold standard).
98
+ os_name: e.g. ``"Linux"``.
99
+ os_version: e.g. ``"Ubuntu 26.04 LTS"``. A drift signal, not identity.
100
+ kernel: The kernel release string. A drift signal, not identity.
101
+ architecture: e.g. ``"x86_64"``.
102
+ cpu_model: The CPU's model string.
103
+ physical_cores: Physical core count.
104
+ logical_cores: Logical core count, hyperthreads included.
105
+ ram_bytes: Total system memory.
106
+ gpus: Every visible GPU. Never summed or averaged by any consumer
107
+ (ADR-0027).
108
+ storage: Attached storage devices. Provenance only; excluded from the fingerprint.
109
+ python_version: The interpreter that produced the measurement — application environment,
110
+ not machine identity.
111
+ observed_at: When this snapshot was taken.
112
+ """
113
+
114
+ machine_fingerprint: str = Field(min_length=1)
115
+ hostname: str | None
116
+ os_name: str | None
117
+ os_version: str | None
118
+ kernel: str | None
119
+ architecture: str | None
120
+ cpu_model: str | None
121
+ physical_cores: MeasurementField = UNSUPPORTED
122
+ logical_cores: MeasurementField = UNSUPPORTED
123
+ ram_bytes: MeasurementField = UNSUPPORTED
124
+ gpus: WireSequence[GpuProfileFields] = ()
125
+ storage: WireSequence[StorageDeviceFields] = ()
126
+ python_version: str | None = None
127
+ observed_at: TimestampField | None = None
128
+
129
+
130
+ MachineProfileOut, MachineProfileIn = payload_models(MachineProfileFields)
131
+ """The ``machine.profile`` payload pair: ``Out`` for writers, ``In`` for readers."""