setspec 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {setspec-0.2.0 → setspec-0.3.0}/.github/workflows/ci.yml +53 -0
- {setspec-0.2.0 → setspec-0.3.0}/CHANGELOG.md +93 -0
- setspec-0.3.0/PHASE4_ISSUES.md +99 -0
- {setspec-0.2.0 → setspec-0.3.0}/PKG-INFO +14 -8
- {setspec-0.2.0 → setspec-0.3.0}/README.md +11 -7
- {setspec-0.2.0 → setspec-0.3.0}/docs/README.md +1 -0
- {setspec-0.2.0 → setspec-0.3.0}/docs/packages/setspec/spec.md +5 -3
- setspec-0.3.0/docs/schemas.md +195 -0
- {setspec-0.2.0 → setspec-0.3.0}/pyproject.toml +7 -0
- {setspec-0.2.0 → setspec-0.3.0}/requirements/ci.lock +145 -0
- {setspec-0.2.0 → setspec-0.3.0}/src/setspec/__about__.py +1 -1
- {setspec-0.2.0 → setspec-0.3.0}/src/setspec/__init__.py +17 -0
- setspec-0.3.0/src/setspec/artifacts.py +341 -0
- {setspec-0.2.0 → setspec-0.3.0}/src/setspec/benchmark/v1.py +6 -5
- {setspec-0.2.0 → setspec-0.3.0}/src/setspec/capability/v1.py +4 -4
- {setspec-0.2.0 → setspec-0.3.0}/src/setspec/envelope.py +22 -29
- {setspec-0.2.0 → setspec-0.3.0}/src/setspec/goal/v1.py +4 -4
- setspec-0.3.0/src/setspec/goldens/benchmark.calibration_report/1.0/full.json +58 -0
- setspec-0.3.0/src/setspec/goldens/benchmark.calibration_report/1.0/gate_failed.json +41 -0
- setspec-0.3.0/src/setspec/goldens/benchmark.calibration_report/1.0/minimal.json +13 -0
- setspec-0.3.0/src/setspec/goldens/benchmark.evidence_bundle/1.0/full.json +218 -0
- setspec-0.3.0/src/setspec/goldens/benchmark.evidence_bundle/1.0/minimal.json +4 -0
- setspec-0.3.0/src/setspec/goldens/benchmark.evidence_bundle/1.0/unsupported.json +67 -0
- setspec-0.3.0/src/setspec/goldens/benchmark.goal_pack/1.0/full.json +69 -0
- setspec-0.3.0/src/setspec/goldens/benchmark.goal_pack/1.0/minimal.json +25 -0
- setspec-0.3.0/src/setspec/goldens/benchmark.goal_pack/1.0/starter_unforked.json +37 -0
- setspec-0.3.0/src/setspec/goldens/benchmark.result/1.0/full.json +208 -0
- setspec-0.3.0/src/setspec/goldens/benchmark.result/1.0/minimal.json +52 -0
- setspec-0.3.0/src/setspec/goldens/benchmark.result/1.0/unsupported.json +129 -0
- setspec-0.3.0/src/setspec/goldens/benchmark.run_summary/1.0/full.json +130 -0
- setspec-0.3.0/src/setspec/goldens/benchmark.run_summary/1.0/minimal.json +33 -0
- setspec-0.3.0/src/setspec/goldens/benchmark.run_summary/1.0/unsupported.json +80 -0
- setspec-0.3.0/src/setspec/goldens/capability.evidence/1.0/full.json +83 -0
- setspec-0.3.0/src/setspec/goldens/capability.evidence/1.0/goal.json +104 -0
- setspec-0.3.0/src/setspec/goldens/capability.evidence/1.0/minimal.json +25 -0
- setspec-0.3.0/src/setspec/goldens/capability.evidence/1.0/unsupported.json +61 -0
- setspec-0.3.0/src/setspec/goldens/machine.profile/1.0/full.json +44 -0
- setspec-0.3.0/src/setspec/goldens/machine.profile/1.0/minimal.json +9 -0
- setspec-0.3.0/src/setspec/goldens/machine.profile/1.0/unsupported.json +27 -0
- setspec-0.3.0/src/setspec/goldens/model.identity/1.0/full.json +34 -0
- setspec-0.3.0/src/setspec/goldens/model.identity/1.0/minimal.json +7 -0
- setspec-0.3.0/src/setspec/goldens/model.identity/1.0/unsupported.json +27 -0
- {setspec-0.2.0 → setspec-0.3.0}/src/setspec/machine/v1.py +1 -1
- {setspec-0.2.0 → setspec-0.3.0}/src/setspec/model/v1.py +9 -6
- setspec-0.3.0/src/setspec/schemas/benchmark.calibration_report/1.0.json +264 -0
- setspec-0.3.0/src/setspec/schemas/benchmark.evidence_bundle/1.0.json +724 -0
- setspec-0.3.0/src/setspec/schemas/benchmark.goal_pack/1.0.json +267 -0
- setspec-0.3.0/src/setspec/schemas/benchmark.result/1.0.json +1358 -0
- setspec-0.3.0/src/setspec/schemas/benchmark.run_summary/1.0.json +819 -0
- setspec-0.3.0/src/setspec/schemas/capability.evidence/1.0.json +693 -0
- setspec-0.3.0/src/setspec/schemas/machine.profile/1.0.json +323 -0
- setspec-0.3.0/src/setspec/schemas/model.identity/1.0.json +299 -0
- setspec-0.3.0/tests/contract/test_cross_version.py +199 -0
- setspec-0.3.0/tests/contract/test_goldens.py +299 -0
- setspec-0.3.0/tests/contract/test_schema_snapshots.py +261 -0
- {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_envelope.py +10 -5
- {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_payloads_goal.py +3 -3
- setspec-0.2.0/src/setspec/artifacts.py +0 -4
- setspec-0.2.0/src/setspec/goldens/.gitkeep +0 -0
- setspec-0.2.0/src/setspec/schemas/.gitkeep +0 -0
- setspec-0.2.0/tests/contract/test_cross_version.py +0 -4
- setspec-0.2.0/tests/contract/test_goldens.py +0 -4
- setspec-0.2.0/tests/contract/test_schema_snapshots.py +0 -4
- {setspec-0.2.0 → setspec-0.3.0}/.editorconfig +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/.github/workflows/release.yml +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/.gitignore +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/.importlinter +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/.pre-commit-config.yaml +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/CONTRIBUTING.md +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/LICENSE +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/SECURITY.md +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/docs/packages/setspec/development-plan.md +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/requirements/README.md +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/requirements/release.in +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/requirements/release.lock +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/src/setspec/base.py +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/src/setspec/error/v1.py +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/src/setspec/errors.py +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/src/setspec/event/v1.py +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/src/setspec/metrics.py +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/src/setspec/provenance.py +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/src/setspec/py.typed +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/src/setspec/serialization.py +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/src/setspec/vocabulary.py +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/tests/conftest.py +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/tests/contract/test_version_negotiation.py +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_base.py +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_errors.py +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_events.py +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_metrics.py +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_payloads_benchmark.py +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_payloads_capability.py +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_serialization.py +0 -0
- {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_vocabulary.py +0 -0
|
@@ -91,6 +91,44 @@ jobs:
|
|
|
91
91
|
- run: pip install . --no-deps
|
|
92
92
|
- run: pytest -m contract
|
|
93
93
|
|
|
94
|
+
# The three published guarantees get their own jobs rather than sharing `contracts`' single
|
|
95
|
+
# red mark, because each answers a different question and the answers are acted on differently:
|
|
96
|
+
# a snapshot diff means "bump the version", a golden failure means "the artifact is wrong", and
|
|
97
|
+
# a cross-version failure means "a consumer at another version just broke".
|
|
98
|
+
schema-snapshot-diff:
|
|
99
|
+
runs-on: ubuntu-latest
|
|
100
|
+
steps:
|
|
101
|
+
- uses: actions/checkout@v4
|
|
102
|
+
- uses: actions/setup-python@v5
|
|
103
|
+
with: { python-version: "3.12" }
|
|
104
|
+
- run: pip install --require-hashes -r requirements/ci.lock
|
|
105
|
+
- run: pip install . --no-deps
|
|
106
|
+
# ADR-0009 rule 7: a schema change without a version bump fails CI. The test regenerates
|
|
107
|
+
# every published schema from its writer model and diffs it against the committed snapshot.
|
|
108
|
+
- run: pytest tests/contract/test_schema_snapshots.py -m contract
|
|
109
|
+
|
|
110
|
+
golden-validation:
|
|
111
|
+
runs-on: ubuntu-latest
|
|
112
|
+
steps:
|
|
113
|
+
- uses: actions/checkout@v4
|
|
114
|
+
- uses: actions/setup-python@v5
|
|
115
|
+
with: { python-version: "3.12" }
|
|
116
|
+
- run: pip install --require-hashes -r requirements/ci.lock
|
|
117
|
+
- run: pip install . --no-deps
|
|
118
|
+
- run: pytest tests/contract/test_goldens.py -m contract
|
|
119
|
+
|
|
120
|
+
cross-version-compatibility:
|
|
121
|
+
runs-on: ubuntu-latest
|
|
122
|
+
steps:
|
|
123
|
+
- uses: actions/checkout@v4
|
|
124
|
+
- uses: actions/setup-python@v5
|
|
125
|
+
with: { python-version: "3.12" }
|
|
126
|
+
- run: pip install --require-hashes -r requirements/ci.lock
|
|
127
|
+
- run: pip install . --no-deps
|
|
128
|
+
# Testing Standards §8.3, proven by the repository that makes the promise: a v1.0 reader
|
|
129
|
+
# accepts a synthetic v1.1 golden without loss and refuses a synthetic v2.0 by major.
|
|
130
|
+
- run: pytest tests/contract/test_cross_version.py -m contract
|
|
131
|
+
|
|
94
132
|
security:
|
|
95
133
|
runs-on: ubuntu-latest
|
|
96
134
|
steps:
|
|
@@ -148,3 +186,18 @@ jobs:
|
|
|
148
186
|
marker = pathlib.Path(setspec.__file__).parent / 'py.typed'
|
|
149
187
|
assert marker.is_file(), 'py.typed missing from the installed wheel'
|
|
150
188
|
"
|
|
189
|
+
# The schemas and goldens are package data (ADR-0009 rule 7), and package data is exactly
|
|
190
|
+
# what a build is most likely to leave out silently: every test in this repository passes
|
|
191
|
+
# against the source tree whether or not the files reach the wheel. This is the only job
|
|
192
|
+
# that would notice, which is why it asserts against the *installed* distribution.
|
|
193
|
+
- name: schemas and goldens ship in the wheel
|
|
194
|
+
run: |
|
|
195
|
+
python -c "
|
|
196
|
+
from setspec import PUBLISHED_SCHEMAS, golden_payloads, json_schema_for
|
|
197
|
+
for schema, versions in PUBLISHED_SCHEMAS.items():
|
|
198
|
+
for version in versions:
|
|
199
|
+
assert json_schema_for(schema, version)['title'] == schema
|
|
200
|
+
goldens = golden_payloads(schema, version)
|
|
201
|
+
assert len(goldens) >= 3, f'{schema} {version} ships {len(goldens)} goldens'
|
|
202
|
+
print(f'{len(PUBLISHED_SCHEMAS)} schemas with goldens loaded from the installed wheel')
|
|
203
|
+
"
|
|
@@ -7,7 +7,84 @@ packaging and release standards §3.
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.3.0] — 2026-08-28
|
|
11
|
+
|
|
10
12
|
### Added
|
|
13
|
+
|
|
14
|
+
- **The v1.0 freeze.** `DRAFT_SCHEMAS` is now empty. Every registered payload type —
|
|
15
|
+
`model.identity`, `machine.profile`, `benchmark.result`, `benchmark.run_summary`,
|
|
16
|
+
`capability.evidence`, `benchmark.evidence_bundle`, `benchmark.goal_pack` and
|
|
17
|
+
`benchmark.calibration_report` — is a published `1.0` that may no longer be reshaped in place.
|
|
18
|
+
FreeWeight's real output (P6 onward, and P8A–8B for the two goal payloads) demanded no field
|
|
19
|
+
change, so the freeze is a promotion rather than a correction; the one field the pass did add,
|
|
20
|
+
`MetricValueFields.metric_key`, landed before it and is frozen *with* the schema.
|
|
21
|
+
|
|
22
|
+
The set survives its own emptiness deliberately: it is the mechanism a future draft uses, and
|
|
23
|
+
deleting it would leave the next provisional payload type either shipping silently provisional or
|
|
24
|
+
inventing a second way to say so.
|
|
25
|
+
|
|
26
|
+
- **Generated JSON Schema for every published version**, committed as package data under
|
|
27
|
+
`setspec/schemas/<schema>/<version>.json` and reachable through `json_schema_for()`. Generated
|
|
28
|
+
from the **writer** (`extra="forbid"`) half, so a published document says
|
|
29
|
+
`additionalProperties: false` — which is the contract API standards §7 rule 5 states and the one
|
|
30
|
+
a producer's test suite asserts its output against. The reader policy stays where it belongs, in
|
|
31
|
+
the reader: `load_envelope` accepts an unknown minor by comparing majors, never by loosening a
|
|
32
|
+
version's schema to admit fields it cannot describe.
|
|
33
|
+
|
|
34
|
+
Every `$ref` resolves inside the document's own `$defs`. Nothing is fetched (spec §14).
|
|
35
|
+
|
|
36
|
+
- **25 golden payloads**, three or more per version, under
|
|
37
|
+
`setspec/goldens/<schema>/<version>/<name>.json` and reachable through `golden_payloads()` and
|
|
38
|
+
`golden_names()`. Each version ships a `minimal` (only required fields), a `full` (every field
|
|
39
|
+
populated) and, wherever the payload carries a measurement at all, an `unsupported`-heavy example.
|
|
40
|
+
`capability.evidence` adds `goal`, a calibrated `user.noir_tech_voice` record with the whole
|
|
41
|
+
goal-sourced group populated; `benchmark.goal_pack` adds `starter_unforked`; and
|
|
42
|
+
`benchmark.calibration_report` adds `gate_failed` — the outcome `capability.evidence`
|
|
43
|
+
deliberately cannot express, because a goal below its gate emits no evidence record at all.
|
|
44
|
+
|
|
45
|
+
Which versions need an `unsupported` golden is derived from the published schema rather than
|
|
46
|
+
listed in the test: a payload whose document contains no `{"const": "unsupported"}` branch has no
|
|
47
|
+
measurement field to leave unsupported, so `benchmark.goal_pack` and
|
|
48
|
+
`benchmark.calibration_report` are excused by the artifact itself rather than by an exception
|
|
49
|
+
someone maintains.
|
|
50
|
+
|
|
51
|
+
- **`setspec.artifacts`** — `PUBLISHED_SCHEMAS`, `json_schema_for()`, `golden_payloads()`,
|
|
52
|
+
`golden_names()`, `payload_pair()`, `build_json_schema()` and `render_schema_document()`. The
|
|
53
|
+
first four are re-exported from `setspec` directly, matching spec §7's "schema artefacts" group:
|
|
54
|
+
an accessor that *returns* a version's artifacts is not itself versioned.
|
|
55
|
+
|
|
56
|
+
`PUBLISHED_SCHEMAS` records **exact** versions, unlike `SUPPORTED_SCHEMAS`, which records
|
|
57
|
+
supported majors. An artifact is a file that either exists or does not, so "highest minor known"
|
|
58
|
+
is the wrong shape for it — and a contract test asserts the two agree, so a schema that can be
|
|
59
|
+
negotiated but not validated, or validated but not negotiated, fails the build.
|
|
60
|
+
|
|
61
|
+
- **Three contract jobs, one per published guarantee.** `schema-snapshot-diff` regenerates every
|
|
62
|
+
schema from its models and diffs it against the committed snapshot — ADR-0009 rule 7, *a schema
|
|
63
|
+
change without a version bump fails CI*, made mechanical and proven by a deliberate test-only
|
|
64
|
+
mutation. `golden-validation` checks every golden against its writer model, its reader model, the
|
|
65
|
+
published JSON Schema and a canonical round trip. `cross-version-compatibility` builds a synthetic
|
|
66
|
+
`1.1` document from each `full` golden and asserts a `1.0` reader accepts it, preserves the field
|
|
67
|
+
it does not know, and re-exports it intact — then builds a synthetic `2.0` and asserts the refusal
|
|
68
|
+
names both versions. They are separate jobs rather than steps because the three failures are acted
|
|
69
|
+
on differently: bump the version, fix the artifact, or go and warn a consumer.
|
|
70
|
+
|
|
71
|
+
- **`install-check` now asserts the package data reaches the wheel.** Every test in this repository
|
|
72
|
+
passes against the source tree whether or not the schemas and goldens are built into the
|
|
73
|
+
distribution, so this job is the only one that would notice. It loads all eight schemas and their
|
|
74
|
+
goldens out of the *installed* wheel.
|
|
75
|
+
|
|
76
|
+
- **`docs/schemas.md`** — the human-readable catalogue: what ships and where, every payload type
|
|
77
|
+
with its required-field count and its goldens, how to consume the artifacts from a repository that
|
|
78
|
+
shares no code with this one, and a table of every cross-field rule the JSON Schema **cannot**
|
|
79
|
+
express and the models therefore still own.
|
|
80
|
+
|
|
81
|
+
- **`jsonschema` and `types-jsonschema` in the `dev` extra.** Test-only, and justified by the
|
|
82
|
+
assertion they exist for: a golden is validated against the *published document* a non-Python
|
|
83
|
+
consumer receives, not only against the model that generated it. Without a real validator that
|
|
84
|
+
check would be a hand-rolled subset of draft 2020-12 maintained in this repository — the thing
|
|
85
|
+
under test also serving as the thing testing it. Nothing under `src/` imports either, and the base
|
|
86
|
+
install still pulls in no validator. `requirements/ci.lock` regenerated.
|
|
87
|
+
|
|
11
88
|
- **`metric_key` on `MetricValueFields`.** The model declared value, unit, aggregation, direction,
|
|
12
89
|
sample count and dispersion — and nothing saying *which metric it is*. Both
|
|
13
90
|
`BenchmarkResultFields.metrics` and `BenchmarkRunSummaryFields.aggregate_metrics` carry sequences
|
|
@@ -58,6 +135,14 @@ packaging and release standards §3.
|
|
|
58
135
|
|
|
59
136
|
### Changed
|
|
60
137
|
|
|
138
|
+
- The five payload modules' status notes read **frozen (`1.0`)** rather than **draft (`1.0`)**, and
|
|
139
|
+
say what the freeze binds them to: a new optional field is a minor bump, and a removal, rename,
|
|
140
|
+
retype or tightening is a major — never an edit in place.
|
|
141
|
+
- The two tests that asserted the draft state now assert the frozen one
|
|
142
|
+
(`test_no_registered_schema_is_still_draft`, `test_the_schema_is_frozen`). They are the same
|
|
143
|
+
guarantee read from the other side, and leaving them asserting `DRAFT_SCHEMAS == SUPPORTED_SCHEMAS`
|
|
144
|
+
would have made the freeze a change the test suite refused.
|
|
145
|
+
|
|
61
146
|
- A bare reserved root is now refused by `validate_capability` and reported `False` by
|
|
62
147
|
`is_known_capability`. `user` is a namespace, not a capability: a payload claiming
|
|
63
148
|
`capability_id: "user"` has lost the identity that is the entire point of the namespace.
|
|
@@ -71,6 +156,7 @@ packaging and release standards §3.
|
|
|
71
156
|
declaring `1.2` or later is unaffected. Producers on `1.1` must emit roots that exist at `1.1`.
|
|
72
157
|
|
|
73
158
|
### Fixed
|
|
159
|
+
|
|
74
160
|
- **Coverage measured a directory nothing imports.** `[tool.coverage.run] source` named
|
|
75
161
|
`src/setspec`, the source *path*, while CI installs the built distribution — so the moment the
|
|
76
162
|
jobs stopped using an editable install, coverage reported **0 %** and failed the 95 % floor for a
|
|
@@ -89,6 +175,13 @@ packaging and release standards §3.
|
|
|
89
175
|
(`baseaicore`, `modelrack`, `sweatmeter`). Without it, `mypy --strict` in a consuming repository
|
|
90
176
|
cannot see this package's types at all and treats every import from it as untyped.
|
|
91
177
|
|
|
178
|
+
### Known gaps
|
|
179
|
+
|
|
180
|
+
- `event.envelope` and `error.envelope` (Phase 3) and `prompt.record` / `prompt.manifest`
|
|
181
|
+
(Phase 5) are **not** part of this freeze, because they do not exist yet: a schema is frozen by
|
|
182
|
+
being published, and an unwritten one has nothing to publish. The freeze covers the eight payload
|
|
183
|
+
types that cross an application boundary today, not the eleven ADR-0009 anticipates.
|
|
184
|
+
|
|
92
185
|
## [0.2.0] — 2026-08-23
|
|
93
186
|
|
|
94
187
|
Phase 2 of the [development plan](docs/packages/setspec/development-plan.md): provisional
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
# Phase 4 — issues to address
|
|
2
|
+
|
|
3
|
+
Written at the end of Phase 4 (freeze v1.0, publish schemas and goldens; `setspec 0.3.0`). Each
|
|
4
|
+
entry is something a later phase, a docs change, a consumer or a release step has to resolve.
|
|
5
|
+
Nothing here blocks Phase 4's acceptance criteria; everything here would become a defect if it were
|
|
6
|
+
forgotten.
|
|
7
|
+
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
## Status — 2026-08-28
|
|
11
|
+
|
|
12
|
+
| # | Issue | Status |
|
|
13
|
+
|---|---|---|
|
|
14
|
+
| 1 | The freeze covers eight payload types, not ADR-0009's eleven | **Open — by design.** Phases 3 and 5 own the other three. |
|
|
15
|
+
| 2 | Goldens are authored inputs, but a SetSpec writer emits every declared key | **Needs a decision.** |
|
|
16
|
+
| 3 | `ContributingMetricFields.metric_key` carries no pattern while `MetricValueFields.metric_key` does | **Needs a decision.** |
|
|
17
|
+
| 4 | Cross-field rules are invisible in the published JSON Schema | **Open — documented.** `docs/schemas.md` §4 lists every one. |
|
|
18
|
+
| 5 | `jsonschema` joined the `dev` extra; `requirements/ci.lock` regenerated | **Closed** — verify `pip-audit` in CI. |
|
|
19
|
+
| 6 | The release is not tagged or published yet | **Owner: you.** Commands below. |
|
|
20
|
+
|
|
21
|
+
---
|
|
22
|
+
|
|
23
|
+
## 1. The freeze covers eight payload types, not eleven
|
|
24
|
+
|
|
25
|
+
ADR-0009 lists eleven initial payload types. `event.envelope` and `error.envelope` (Phase 3) and
|
|
26
|
+
`prompt.record` / `prompt.manifest` (Phase 5) are not written, so they are not frozen — a schema is
|
|
27
|
+
frozen by being published, and an unwritten one has nothing to publish. `DRAFT_SCHEMAS` is empty
|
|
28
|
+
**now**; when Phase 3 or 5 lands a new payload type it should re-enter that set until its own
|
|
29
|
+
freeze, which is exactly the mechanism the set survives its emptiness for. The changelog's *Known
|
|
30
|
+
gaps* section says the same. Nothing to do until those phases start, except not to read "empty"
|
|
31
|
+
as "finished".
|
|
32
|
+
|
|
33
|
+
## 2. Goldens are authored inputs, but a SetSpec writer emits every declared key
|
|
34
|
+
|
|
35
|
+
`dump_envelope(CapabilityEvidenceOut(...))` serialises through `model_dump()`, which writes every
|
|
36
|
+
declared field — a non-goal record carries `goal_hash: null`, `uncalibrated: false`, and so on. The
|
|
37
|
+
`capability.evidence/1.0/full` golden was authored the other way round: it omits the goal group
|
|
38
|
+
entirely, because "fully populated" was read as "every field a *non-goal* record populates". Both
|
|
39
|
+
forms validate under both models, so the contract is not broken, but a producer's structural test
|
|
40
|
+
("my keys match the full golden's keys") fails against the file and passes against the file *as a
|
|
41
|
+
SetSpec writer would dump it*. FreeWeight's contract test compares against the latter.
|
|
42
|
+
|
|
43
|
+
**Decision needed:** should goldens be committed *as written* (every declared key present, nulls
|
|
44
|
+
included), so that "matches the golden structurally" is a byte-level statement? If yes, regenerate
|
|
45
|
+
the 25 goldens through their `Out` models once and add a contract test that a golden equals its own
|
|
46
|
+
writer-dumped form. If no, say so in `docs/schemas.md` §3 so consumers compare against the dumped
|
|
47
|
+
form, as FreeWeight now does.
|
|
48
|
+
|
|
49
|
+
## 3. `ContributingMetricFields.metric_key` has no pattern
|
|
50
|
+
|
|
51
|
+
`MetricValueFields.metric_key` is constrained to lower snake case, dot-separable
|
|
52
|
+
(`^[a-z][a-z0-9_]*(\.[a-z][a-z0-9_]*)*$`). `ContributingMetricFields.metric_key` — the key inside
|
|
53
|
+
`capability.evidence.contributing_metrics` — is only `min_length=1`. FreeWeight writes
|
|
54
|
+
`<suite_key>.<metric_key>` there (`native.tool_use.task_success`), `criterion.<key>` for a goal's
|
|
55
|
+
own record and `goal.<slug>.composite_score` for a goal contributing to a shipped capability; all
|
|
56
|
+
three happen to satisfy the stricter pattern. Tightening the field is a **major** change now that
|
|
57
|
+
`1.0` is frozen, so it can only land as `capability.evidence 2.0`, and only if a consumer ever
|
|
58
|
+
needs the guarantee. Recorded so the asymmetry is a decision rather than an accident.
|
|
59
|
+
|
|
60
|
+
## 4. Cross-field rules are invisible in the published JSON Schema
|
|
61
|
+
|
|
62
|
+
Pydantic renders types, ranges, patterns and required keys; it cannot render a `model_validator`.
|
|
63
|
+
`runtime_profile_hash` agreeing with its profile, `score_method_mix` summing to one,
|
|
64
|
+
`measured_at ≤ computed_at`, the goal group's five coherence rules — every one is enforced by the
|
|
65
|
+
models only. `docs/schemas.md` §4 tables them, and every golden is validated against *both* the
|
|
66
|
+
schema and the model for exactly this reason. A non-Python consumer that needs those rules needs a
|
|
67
|
+
second validation step this package does not ship. Nothing to do unless such a consumer appears.
|
|
68
|
+
|
|
69
|
+
## 5. `jsonschema` in the `dev` extra
|
|
70
|
+
|
|
71
|
+
Added so the golden contract test validates against the *published* document with a real
|
|
72
|
+
draft-2020-12 validator rather than a hand-rolled subset. Test only — nothing under `src/`
|
|
73
|
+
imports it and the base install pulls in no validator. `requirements/ci.lock` was regenerated with
|
|
74
|
+
`pip-compile --generate-hashes` (jsonschema 4.26.0, jsonschema-specifications, referencing,
|
|
75
|
+
rpds-py, attrs, types-jsonschema). `release.lock` is unchanged. CI's `security` job audits both
|
|
76
|
+
locks; the first run after this lands is the one to watch.
|
|
77
|
+
|
|
78
|
+
## 6. Tagging and publishing 0.3.0
|
|
79
|
+
|
|
80
|
+
Nothing was tagged or published. `__about__.py` says `0.3.0` and the changelog carries a dated
|
|
81
|
+
`[0.3.0]` section. The release procedure (packaging standards §6), from the SetSpec repository:
|
|
82
|
+
|
|
83
|
+
```bash
|
|
84
|
+
cd ~/ai/suite/py/SetSpec
|
|
85
|
+
git add -A
|
|
86
|
+
git commit -m "feat(setspec): freeze v1.0, publish JSON Schema and goldens (Phase 4, 0.3.0)"
|
|
87
|
+
git push origin main
|
|
88
|
+
# wait for CI to be green on main, then:
|
|
89
|
+
git tag -a v0.3.0 -m "setspec 0.3.0 — v1.0 contracts frozen; JSON Schema and goldens published"
|
|
90
|
+
git push origin v0.3.0
|
|
91
|
+
# release.yml builds, tests the wheel, publishes via Trusted Publishing and creates the release.
|
|
92
|
+
# Verify:
|
|
93
|
+
python -m venv /tmp/setspec-check && /tmp/setspec-check/bin/pip install setspec==0.3.0 && \
|
|
94
|
+
/tmp/setspec-check/bin/python -c "from setspec import PUBLISHED_SCHEMAS, golden_payloads, SchemaVersion; \
|
|
95
|
+
print(len(PUBLISHED_SCHEMAS), len(golden_payloads('capability.evidence', SchemaVersion(1, 0))))"
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
FreeWeight's `pyproject.toml` already pins `setspec>=0.3,<0.4`; its `install-check` job cannot
|
|
99
|
+
resolve until this release is on PyPI.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: setspec
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Every versioned data contract that crosses an application boundary: benchmark results, capability evidence, event/error envelopes, prompt records.
|
|
5
5
|
Project-URL: Homepage, https://github.com/JPKell/SetSpec
|
|
6
6
|
Project-URL: Documentation, https://github.com/JPKell/SetSpec/tree/main/docs
|
|
@@ -21,25 +21,31 @@ Requires-Dist: jinja2<4,>=3.1
|
|
|
21
21
|
Requires-Dist: pydantic<3,>=2.9
|
|
22
22
|
Provides-Extra: dev
|
|
23
23
|
Requires-Dist: import-linter<3,>=2.0; extra == 'dev'
|
|
24
|
+
Requires-Dist: jsonschema<5,>=4.23; extra == 'dev'
|
|
24
25
|
Requires-Dist: mypy<2,>=1.11; extra == 'dev'
|
|
25
26
|
Requires-Dist: pytest-cov<6,>=5; extra == 'dev'
|
|
26
27
|
Requires-Dist: pytest-randomly<4,>=3; extra == 'dev'
|
|
27
28
|
Requires-Dist: pytest<10,>=9.0.3; extra == 'dev'
|
|
28
29
|
Requires-Dist: respx<1,>=0.21; extra == 'dev'
|
|
29
30
|
Requires-Dist: ruff<1,>=0.6; extra == 'dev'
|
|
31
|
+
Requires-Dist: types-jsonschema<5,>=4.23; extra == 'dev'
|
|
30
32
|
Description-Content-Type: text/markdown
|
|
31
33
|
|
|
32
34
|
# SetSpec
|
|
33
35
|
|
|
34
36
|
Every versioned data contract that crosses an application boundary: benchmark results, capability evidence, event/error envelopes, prompt records.
|
|
35
37
|
|
|
36
|
-
**Status:** `0.
|
|
37
|
-
are
|
|
38
|
-
`benchmark.run_summary`, `capability.evidence
|
|
39
|
-
`
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
38
|
+
**Status:** `0.3.0` — Phases 1–2, 3A and 4 complete, and **the v1.0 contracts are frozen.** Eight
|
|
39
|
+
payload types are published at `1.0` — `model.identity`, `machine.profile`, `benchmark.result`,
|
|
40
|
+
`benchmark.run_summary`, `capability.evidence`, `benchmark.evidence_bundle`,
|
|
41
|
+
`benchmark.goal_pack` and `benchmark.calibration_report` — each with generated JSON Schema and at
|
|
42
|
+
least three golden payloads shipped as package data. `setspec.DRAFT_SCHEMAS` is empty, which is
|
|
43
|
+
where the freeze is readable at runtime rather than only stated here; from now on a new optional
|
|
44
|
+
field is a minor bump and anything else is a major, enforced by a snapshot diff in CI.
|
|
45
|
+
|
|
46
|
+
The [schema catalogue](docs/schemas.md) lists every payload type, its artifacts, and the
|
|
47
|
+
cross-field rules the JSON Schema cannot express. Event and error envelopes (Phase 3) and prompt
|
|
48
|
+
records (Phase 5) are not yet written and are therefore not part of the freeze. See the
|
|
43
49
|
[development plan](docs/packages/setspec/development-plan.md) for what each phase adds.
|
|
44
50
|
|
|
45
51
|
Part of the **Local AI Suite**.
|
|
@@ -2,13 +2,17 @@
|
|
|
2
2
|
|
|
3
3
|
Every versioned data contract that crosses an application boundary: benchmark results, capability evidence, event/error envelopes, prompt records.
|
|
4
4
|
|
|
5
|
-
**Status:** `0.
|
|
6
|
-
are
|
|
7
|
-
`benchmark.run_summary`, `capability.evidence
|
|
8
|
-
`
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
5
|
+
**Status:** `0.3.0` — Phases 1–2, 3A and 4 complete, and **the v1.0 contracts are frozen.** Eight
|
|
6
|
+
payload types are published at `1.0` — `model.identity`, `machine.profile`, `benchmark.result`,
|
|
7
|
+
`benchmark.run_summary`, `capability.evidence`, `benchmark.evidence_bundle`,
|
|
8
|
+
`benchmark.goal_pack` and `benchmark.calibration_report` — each with generated JSON Schema and at
|
|
9
|
+
least three golden payloads shipped as package data. `setspec.DRAFT_SCHEMAS` is empty, which is
|
|
10
|
+
where the freeze is readable at runtime rather than only stated here; from now on a new optional
|
|
11
|
+
field is a minor bump and anything else is a major, enforced by a snapshot diff in CI.
|
|
12
|
+
|
|
13
|
+
The [schema catalogue](docs/schemas.md) lists every payload type, its artifacts, and the
|
|
14
|
+
cross-field rules the JSON Schema cannot express. Event and error envelopes (Phase 3) and prompt
|
|
15
|
+
records (Phase 5) are not yet written and are therefore not part of the freeze. See the
|
|
12
16
|
[development plan](docs/packages/setspec/development-plan.md) for what each phase adds.
|
|
13
17
|
|
|
14
18
|
Part of the **Local AI Suite**.
|
|
@@ -4,5 +4,6 @@ This directory contains the maintained documentation for SetSpec.
|
|
|
4
4
|
|
|
5
5
|
## Documents
|
|
6
6
|
|
|
7
|
+
- [Schema catalogue](schemas.md) — every payload type and version, with its artifacts
|
|
7
8
|
- [Development plan](packages/setspec/development-plan.md)
|
|
8
9
|
- [Specification](packages/setspec/spec.md)
|
|
@@ -124,9 +124,11 @@ RESERVED_ROOTS: frozenset[str] # roots valid ONLY as a specialization; {"
|
|
|
124
124
|
validate_capability(capability_id: str) -> CapabilityId
|
|
125
125
|
is_known_capability(capability_id: str) -> bool
|
|
126
126
|
|
|
127
|
-
# Schema artefacts
|
|
128
|
-
|
|
129
|
-
|
|
127
|
+
# Schema artefacts (package data; every published version keeps both for the life of its major)
|
|
128
|
+
PUBLISHED_SCHEMAS: Mapping[str, tuple[SchemaVersion, ...]] # exact versions with committed artefacts
|
|
129
|
+
json_schema_for(schema: str, version: SchemaVersion) -> dict # the committed JSON Schema snapshot
|
|
130
|
+
golden_payloads(schema: str, version: SchemaVersion) -> list[dict] # ≥ 3 per version, by name
|
|
131
|
+
golden_names(schema: str, version: SchemaVersion) -> tuple[str, ...] # the same order as above
|
|
130
132
|
|
|
131
133
|
# Prompts (ADR-0028) — one implementation of a determinism contract, not three
|
|
132
134
|
class PromptLibrary:
|
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
# Schema catalogue
|
|
2
|
+
|
|
3
|
+
Every payload type `setspec` publishes, at every version, with the artifacts that make it usable
|
|
4
|
+
from a repository that shares no code with this one.
|
|
5
|
+
|
|
6
|
+
**Status: frozen at `1.0`** (Phase 4, `setspec 0.3.0`). `DRAFT_SCHEMAS` is empty. From here every
|
|
7
|
+
change follows the ordinary rules: a new optional field is a **minor** bump, and a removed,
|
|
8
|
+
renamed, retyped or newly-tightened field is a **major**. Neither happens by editing a payload
|
|
9
|
+
module in place — the snapshot contract test fails the build if the generated schema stops matching
|
|
10
|
+
the committed one, which is ADR-0009 rule 7 made mechanical.
|
|
11
|
+
|
|
12
|
+
---
|
|
13
|
+
|
|
14
|
+
## 1. What ships, and where it lives
|
|
15
|
+
|
|
16
|
+
| Artifact | Location in the installed package | Accessor |
|
|
17
|
+
|---|---|---|
|
|
18
|
+
| JSON Schema, one per version | `setspec/schemas/<schema>/<version>.json` | `json_schema_for(schema, version)` |
|
|
19
|
+
| Golden payloads, ≥ 3 per version | `setspec/goldens/<schema>/<version>/<name>.json` | `golden_payloads(schema, version)` |
|
|
20
|
+
| The names of those goldens | — | `golden_names(schema, version)` |
|
|
21
|
+
| Every published schema and version | — | `PUBLISHED_SCHEMAS` |
|
|
22
|
+
|
|
23
|
+
All four are importable from `setspec` directly. The files are **package data**, loadable through
|
|
24
|
+
`importlib.resources` and present in the built wheel — the `install-check` CI job asserts that
|
|
25
|
+
against the installed distribution, because every test in this repository passes against the source
|
|
26
|
+
tree whether or not the files reach the wheel.
|
|
27
|
+
|
|
28
|
+
Nothing is fetched. A published schema resolves every `$ref` inside its own `$defs`, so validating
|
|
29
|
+
an export never depends on a registry being reachable (spec §14).
|
|
30
|
+
|
|
31
|
+
## 2. The catalogue
|
|
32
|
+
|
|
33
|
+
| Schema | Version | Python module | Writer / reader | Goldens |
|
|
34
|
+
|---|---|---|---|---|
|
|
35
|
+
| `model.identity` | 1.0 | `setspec.model.v1` | `ModelIdentityOut` / `ModelIdentityIn` | `minimal`, `full`, `unsupported` |
|
|
36
|
+
| `machine.profile` | 1.0 | `setspec.machine.v1` | `MachineProfileOut` / `MachineProfileIn` | `minimal`, `full`, `unsupported` |
|
|
37
|
+
| `benchmark.result` | 1.0 | `setspec.benchmark.v1` | `BenchmarkResultOut` / `BenchmarkResultIn` | `minimal`, `full`, `unsupported` |
|
|
38
|
+
| `benchmark.run_summary` | 1.0 | `setspec.benchmark.v1` | `BenchmarkRunSummaryOut` / `BenchmarkRunSummaryIn` | `minimal`, `full`, `unsupported` |
|
|
39
|
+
| `capability.evidence` | 1.0 | `setspec.capability.v1` | `CapabilityEvidenceOut` / `CapabilityEvidenceIn` | `minimal`, `full`, `goal`, `unsupported` |
|
|
40
|
+
| `benchmark.evidence_bundle` | 1.0 | `setspec.capability.v1` | `EvidenceBundleOut` / `EvidenceBundleIn` | `minimal`, `full`, `unsupported` |
|
|
41
|
+
| `benchmark.goal_pack` | 1.0 | `setspec.goal.v1` | `GoalPackOut` / `GoalPackIn` | `minimal`, `full`, `starter_unforked` |
|
|
42
|
+
| `benchmark.calibration_report` | 1.0 | `setspec.goal.v1` | `CalibrationReportOut` / `CalibrationReportIn` | `minimal`, `full`, `gate_failed` |
|
|
43
|
+
|
|
44
|
+
### `model.identity` — which weights, plus what the provider says about them
|
|
45
|
+
|
|
46
|
+
5 required fields of 25. The identity triple (`provider_kind`, `provider_model_name`,
|
|
47
|
+
`artifact_digest`) plus its two derived fields, then the refreshable descriptor. `canonical_id` and
|
|
48
|
+
`identity_confidence` are **recomputed and checked** against the triple on validation: they are
|
|
49
|
+
pure functions of it, carried on the wire for convenience rather than as independent facts
|
|
50
|
+
(ADR-0024). Every descriptor quantity defaults to `"unsupported"`, because a provider that reports
|
|
51
|
+
no layer count has not reported zero layers.
|
|
52
|
+
|
|
53
|
+
### `machine.profile` — where a measurement happened
|
|
54
|
+
|
|
55
|
+
7 required of 14. A field-for-field mirror of `baseaicore.MachineProfile`, including which fields
|
|
56
|
+
are required-but-nullable: a producer must say it could not read the hostname rather than omit the
|
|
57
|
+
key. `machine_fingerprint` is the one hash-shaped field this package carries **without**
|
|
58
|
+
recomputing it — the policy deciding which fields feed a fingerprint may change, while a profile
|
|
59
|
+
read back years later must still reconstruct exactly as stored.
|
|
60
|
+
|
|
61
|
+
### `benchmark.result` — one benchmark, one subject, one machine
|
|
62
|
+
|
|
63
|
+
12 required of 23, the largest payload in the suite. Carries Machine Identity §6's minimum
|
|
64
|
+
provenance set as nested objects: `suite`, `execution`, `environment`, `application`,
|
|
65
|
+
`reproducibility`, plus the optional `machine_profile` and `telemetry_summary`.
|
|
66
|
+
`runtime_profile_hash` is recomputed from the embedded `runtime_profile` and must agree
|
|
67
|
+
(ADR-0023). Four further cross-field rules apply — see §4.
|
|
68
|
+
|
|
69
|
+
### `benchmark.run_summary` — the roll-up of many results
|
|
70
|
+
|
|
71
|
+
10 required of 15. Deliberately lighter than a result: the same subject/suite/environment building
|
|
72
|
+
blocks, a run state, three timestamps and `aggregate_metrics`. `INTERRUPTED` is a distinct status
|
|
73
|
+
from `FAILED` and means the process died and the run is resumable, not that it failed.
|
|
74
|
+
|
|
75
|
+
### `capability.evidence` — the record LoadCoach routes on
|
|
76
|
+
|
|
77
|
+
14 required of 26, and the suite's most load-bearing contract (ADR-0022). Identity, score,
|
|
78
|
+
confidence, the counts behind them, the hard-separation inputs (`benchmark_versions`,
|
|
79
|
+
`dataset_hashes`, `prompt_subset_hashes`, `goal_hash`, `judge_set`) and the two timestamps whose
|
|
80
|
+
order is enforced: `measured_at` is what freshness decays from and can never follow `computed_at`.
|
|
81
|
+
|
|
82
|
+
The **goal-sourced group** (ADR-0032 §5) is optional and absent on an ordinary record:
|
|
83
|
+
`judge_validity_factor` (exactly `1.0` for every rung 1–4 measurement), `goal_hash`,
|
|
84
|
+
`goal_pack_version`, `score_method_mix`, `judge_set`, `calibration`, `uncalibrated`. The `goal`
|
|
85
|
+
golden populates all of it; `uncalibrated: true` is **refused** rather than published, because a
|
|
86
|
+
goal below its calibration gate emits no record at all.
|
|
87
|
+
|
|
88
|
+
### `benchmark.evidence_bundle` — the FreeWeight → LoadCoach payload
|
|
89
|
+
|
|
90
|
+
2 required of 3: `source_id`, `complete`, `evidence`. `complete: true` is what lets a consumer
|
|
91
|
+
infer removal — evidence held locally for this `source_id` and absent from a complete bundle is
|
|
92
|
+
marked superseded, never deleted, and never inferred from a partial bundle. The bundle carries no
|
|
93
|
+
`generated_at`: that lives on the envelope, and a client stores *that* value to send back as its
|
|
94
|
+
next `?since=`.
|
|
95
|
+
|
|
96
|
+
### `benchmark.goal_pack` — a user-authored goal, portable and hash-pinned
|
|
97
|
+
|
|
98
|
+
6 required of 12. Criteria with their ladder rung and weight, tasks with their prompt hashes, and
|
|
99
|
+
the jury when any criterion is judged. Weights must sum to `1`; a judged criterion needs an
|
|
100
|
+
anchored ordinal scale and a `judge_set`. The author's calibration **grades do not travel** — an
|
|
101
|
+
importer who wants the rubric held to their own taste recalibrates against their own grades.
|
|
102
|
+
|
|
103
|
+
### `benchmark.calibration_report` — how well the judge agreed with the author
|
|
104
|
+
|
|
105
|
+
11 required of 14. A payload in its own right rather than a field group on evidence, because it is
|
|
106
|
+
meaningful precisely when **no** evidence was emitted: `gate_failed` is the golden that shows the
|
|
107
|
+
outcome `capability.evidence` deliberately cannot express (ADR-0032 §3). The verdict must follow
|
|
108
|
+
from the numbers — a `passed_gate` that contradicts `weighted_kappa_w` against `min_agreement` is
|
|
109
|
+
refused.
|
|
110
|
+
|
|
111
|
+
## 3. Consuming these from another repository
|
|
112
|
+
|
|
113
|
+
No shared code, no import of the producing application, no database:
|
|
114
|
+
|
|
115
|
+
```python
|
|
116
|
+
from setspec import SchemaVersion, golden_payloads, json_schema_for, load_envelope
|
|
117
|
+
from setspec.capability.v1 import EvidenceBundleIn
|
|
118
|
+
|
|
119
|
+
# The contract test: read every golden this build publishes.
|
|
120
|
+
for payload in golden_payloads("benchmark.evidence_bundle", SchemaVersion(1, 0)):
|
|
121
|
+
EvidenceBundleIn.model_validate(payload)
|
|
122
|
+
|
|
123
|
+
# The real thing: a file a producer exported.
|
|
124
|
+
envelope = load_envelope(exported_bytes, expect="benchmark.evidence_bundle")
|
|
125
|
+
bundle = EvidenceBundleIn.model_validate(envelope.payload)
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
Read with the **`In`** model, always. It preserves fields this build has not heard of, so an older
|
|
129
|
+
consumer in the middle of a pipeline is a relay rather than a sink (ADR-0009 rule 4). Write with the
|
|
130
|
+
**`Out`** model, which refuses a field the schema does not declare (rule 5).
|
|
131
|
+
|
|
132
|
+
A non-Python consumer validates against `json_schema_for(...)`, or against the file directly:
|
|
133
|
+
|
|
134
|
+
```
|
|
135
|
+
setspec/schemas/benchmark.evidence_bundle/1.0.json
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
The documents declare `$schema: https://json-schema.org/draft/2020-12/schema`.
|
|
139
|
+
|
|
140
|
+
## 4. What the JSON Schema does not say
|
|
141
|
+
|
|
142
|
+
The published documents are generated from the **writer** model, so they carry
|
|
143
|
+
`additionalProperties: false` at the top level — that is the writer's contract, and a producer's
|
|
144
|
+
test suite asserts its output validates against it (testing standards §8.2). The reader policy is a
|
|
145
|
+
*reader behaviour*: `load_envelope` accepts any minor within a supported major, including one it has
|
|
146
|
+
never heard of. That is not a loosening of any one version's schema. A `1.1` document is described
|
|
147
|
+
by the `1.1` schema; a `1.0` reader accepts it by comparing majors.
|
|
148
|
+
|
|
149
|
+
Pydantic renders types, ranges, patterns and required keys. It cannot render a `model_validator`, so
|
|
150
|
+
none of the following appears in the published documents and all of them are enforced by the models:
|
|
151
|
+
|
|
152
|
+
| Payload | Rule the schema cannot state |
|
|
153
|
+
|---|---|
|
|
154
|
+
| `metric.value` (nested) | A real value needs ≥ 1 supported sample; an unsupported value needs 0; dispersion needs ≥ 2 |
|
|
155
|
+
| `model.identity` | `canonical_id` and `identity_confidence` must agree with the identity triple |
|
|
156
|
+
| `benchmark.result` | `runtime_profile_hash` must recompute from `runtime_profile`; `completed_at ≥ started_at`; `completed_cases ≤ total_cases`; `skip_reason` iff skipped; a completed result has metrics |
|
|
157
|
+
| `benchmark.run_summary` | `runtime_profile_hash` agreement; timing order |
|
|
158
|
+
| `capability.evidence` | `capability_id` in the vocabulary; `measured_at ≤ computed_at`; the five goal-group coherence rules; `score_method_mix` sums to 1 over known rungs |
|
|
159
|
+
| `benchmark.goal_pack` | Criterion weights sum to 1; no duplicate keys; rung-appropriate fields; a judged criterion needs a jury |
|
|
160
|
+
| `benchmark.calibration_report` | The gate verdict must follow from `weighted_kappa_w` against `min_agreement` |
|
|
161
|
+
|
|
162
|
+
This is why every golden is validated **twice** — once against the schema a non-Python consumer
|
|
163
|
+
uses, once against the model — and why a consumer that can run Python should use the model.
|
|
164
|
+
|
|
165
|
+
## 5. Versioning and regeneration
|
|
166
|
+
|
|
167
|
+
A schema version is `MAJOR.MINOR`, independent of this package's own version and of any HTTP API
|
|
168
|
+
version. `SUPPORTED_SCHEMAS` declares supported **majors**; `PUBLISHED_SCHEMAS` declares the exact
|
|
169
|
+
versions that have artifacts. A contract test asserts the two agree, so a schema that can be
|
|
170
|
+
negotiated but not validated — or validated but not negotiated — fails the build.
|
|
171
|
+
|
|
172
|
+
Adding a version means adding a module (`setspec.benchmark.v2`), registering it, and committing its
|
|
173
|
+
schema and goldens. The old version stays importable for at least one minor release of every
|
|
174
|
+
consumer (ADR-0009 rule 6).
|
|
175
|
+
|
|
176
|
+
Regenerate the snapshots after any deliberate model change:
|
|
177
|
+
|
|
178
|
+
```bash
|
|
179
|
+
python - <<'PY'
|
|
180
|
+
from pathlib import Path
|
|
181
|
+
from setspec.artifacts import PUBLISHED_SCHEMAS, build_json_schema, render_schema_document
|
|
182
|
+
|
|
183
|
+
for schema, versions in PUBLISHED_SCHEMAS.items():
|
|
184
|
+
for version in versions:
|
|
185
|
+
path = Path("src/setspec/schemas", schema, f"{version}.json")
|
|
186
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
187
|
+
path.write_text(
|
|
188
|
+
render_schema_document(build_json_schema(schema, version)), encoding="utf-8"
|
|
189
|
+
)
|
|
190
|
+
PY
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
Then answer the question CI just asked: was that change additive, or did it deserve a major bump?
|
|
194
|
+
Goldens are **not** regenerated. They are authored, committed and left alone — a golden that is
|
|
195
|
+
rebuilt from the model whenever the model changes is not a golden, it is a mirror.
|
|
@@ -36,6 +36,13 @@ dev = [
|
|
|
36
36
|
"ruff>=0.6,<1",
|
|
37
37
|
"import-linter>=2.0,<3",
|
|
38
38
|
"respx>=0.21,<1",
|
|
39
|
+
# Phase 4's contract tests validate every golden against the *published* JSON Schema, not only
|
|
40
|
+
# against the model that generated it. Without a real validator that assertion would be a
|
|
41
|
+
# hand-rolled subset of draft 2020-12 written in this repository — which is to say, the thing
|
|
42
|
+
# under test would also be the thing testing it. Test-only: nothing under `src/` imports it,
|
|
43
|
+
# and `setspec` itself pulls in no validator (packaging standards §4).
|
|
44
|
+
"jsonschema>=4.23,<5",
|
|
45
|
+
"types-jsonschema>=4.23,<5",
|
|
39
46
|
]
|
|
40
47
|
|
|
41
48
|
[project.urls]
|