setspec 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (94) hide show
  1. {setspec-0.2.0 → setspec-0.3.0}/.github/workflows/ci.yml +53 -0
  2. {setspec-0.2.0 → setspec-0.3.0}/CHANGELOG.md +93 -0
  3. setspec-0.3.0/PHASE4_ISSUES.md +99 -0
  4. {setspec-0.2.0 → setspec-0.3.0}/PKG-INFO +14 -8
  5. {setspec-0.2.0 → setspec-0.3.0}/README.md +11 -7
  6. {setspec-0.2.0 → setspec-0.3.0}/docs/README.md +1 -0
  7. {setspec-0.2.0 → setspec-0.3.0}/docs/packages/setspec/spec.md +5 -3
  8. setspec-0.3.0/docs/schemas.md +195 -0
  9. {setspec-0.2.0 → setspec-0.3.0}/pyproject.toml +7 -0
  10. {setspec-0.2.0 → setspec-0.3.0}/requirements/ci.lock +145 -0
  11. {setspec-0.2.0 → setspec-0.3.0}/src/setspec/__about__.py +1 -1
  12. {setspec-0.2.0 → setspec-0.3.0}/src/setspec/__init__.py +17 -0
  13. setspec-0.3.0/src/setspec/artifacts.py +341 -0
  14. {setspec-0.2.0 → setspec-0.3.0}/src/setspec/benchmark/v1.py +6 -5
  15. {setspec-0.2.0 → setspec-0.3.0}/src/setspec/capability/v1.py +4 -4
  16. {setspec-0.2.0 → setspec-0.3.0}/src/setspec/envelope.py +22 -29
  17. {setspec-0.2.0 → setspec-0.3.0}/src/setspec/goal/v1.py +4 -4
  18. setspec-0.3.0/src/setspec/goldens/benchmark.calibration_report/1.0/full.json +58 -0
  19. setspec-0.3.0/src/setspec/goldens/benchmark.calibration_report/1.0/gate_failed.json +41 -0
  20. setspec-0.3.0/src/setspec/goldens/benchmark.calibration_report/1.0/minimal.json +13 -0
  21. setspec-0.3.0/src/setspec/goldens/benchmark.evidence_bundle/1.0/full.json +218 -0
  22. setspec-0.3.0/src/setspec/goldens/benchmark.evidence_bundle/1.0/minimal.json +4 -0
  23. setspec-0.3.0/src/setspec/goldens/benchmark.evidence_bundle/1.0/unsupported.json +67 -0
  24. setspec-0.3.0/src/setspec/goldens/benchmark.goal_pack/1.0/full.json +69 -0
  25. setspec-0.3.0/src/setspec/goldens/benchmark.goal_pack/1.0/minimal.json +25 -0
  26. setspec-0.3.0/src/setspec/goldens/benchmark.goal_pack/1.0/starter_unforked.json +37 -0
  27. setspec-0.3.0/src/setspec/goldens/benchmark.result/1.0/full.json +208 -0
  28. setspec-0.3.0/src/setspec/goldens/benchmark.result/1.0/minimal.json +52 -0
  29. setspec-0.3.0/src/setspec/goldens/benchmark.result/1.0/unsupported.json +129 -0
  30. setspec-0.3.0/src/setspec/goldens/benchmark.run_summary/1.0/full.json +130 -0
  31. setspec-0.3.0/src/setspec/goldens/benchmark.run_summary/1.0/minimal.json +33 -0
  32. setspec-0.3.0/src/setspec/goldens/benchmark.run_summary/1.0/unsupported.json +80 -0
  33. setspec-0.3.0/src/setspec/goldens/capability.evidence/1.0/full.json +83 -0
  34. setspec-0.3.0/src/setspec/goldens/capability.evidence/1.0/goal.json +104 -0
  35. setspec-0.3.0/src/setspec/goldens/capability.evidence/1.0/minimal.json +25 -0
  36. setspec-0.3.0/src/setspec/goldens/capability.evidence/1.0/unsupported.json +61 -0
  37. setspec-0.3.0/src/setspec/goldens/machine.profile/1.0/full.json +44 -0
  38. setspec-0.3.0/src/setspec/goldens/machine.profile/1.0/minimal.json +9 -0
  39. setspec-0.3.0/src/setspec/goldens/machine.profile/1.0/unsupported.json +27 -0
  40. setspec-0.3.0/src/setspec/goldens/model.identity/1.0/full.json +34 -0
  41. setspec-0.3.0/src/setspec/goldens/model.identity/1.0/minimal.json +7 -0
  42. setspec-0.3.0/src/setspec/goldens/model.identity/1.0/unsupported.json +27 -0
  43. {setspec-0.2.0 → setspec-0.3.0}/src/setspec/machine/v1.py +1 -1
  44. {setspec-0.2.0 → setspec-0.3.0}/src/setspec/model/v1.py +9 -6
  45. setspec-0.3.0/src/setspec/schemas/benchmark.calibration_report/1.0.json +264 -0
  46. setspec-0.3.0/src/setspec/schemas/benchmark.evidence_bundle/1.0.json +724 -0
  47. setspec-0.3.0/src/setspec/schemas/benchmark.goal_pack/1.0.json +267 -0
  48. setspec-0.3.0/src/setspec/schemas/benchmark.result/1.0.json +1358 -0
  49. setspec-0.3.0/src/setspec/schemas/benchmark.run_summary/1.0.json +819 -0
  50. setspec-0.3.0/src/setspec/schemas/capability.evidence/1.0.json +693 -0
  51. setspec-0.3.0/src/setspec/schemas/machine.profile/1.0.json +323 -0
  52. setspec-0.3.0/src/setspec/schemas/model.identity/1.0.json +299 -0
  53. setspec-0.3.0/tests/contract/test_cross_version.py +199 -0
  54. setspec-0.3.0/tests/contract/test_goldens.py +299 -0
  55. setspec-0.3.0/tests/contract/test_schema_snapshots.py +261 -0
  56. {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_envelope.py +10 -5
  57. {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_payloads_goal.py +3 -3
  58. setspec-0.2.0/src/setspec/artifacts.py +0 -4
  59. setspec-0.2.0/src/setspec/goldens/.gitkeep +0 -0
  60. setspec-0.2.0/src/setspec/schemas/.gitkeep +0 -0
  61. setspec-0.2.0/tests/contract/test_cross_version.py +0 -4
  62. setspec-0.2.0/tests/contract/test_goldens.py +0 -4
  63. setspec-0.2.0/tests/contract/test_schema_snapshots.py +0 -4
  64. {setspec-0.2.0 → setspec-0.3.0}/.editorconfig +0 -0
  65. {setspec-0.2.0 → setspec-0.3.0}/.github/workflows/release.yml +0 -0
  66. {setspec-0.2.0 → setspec-0.3.0}/.gitignore +0 -0
  67. {setspec-0.2.0 → setspec-0.3.0}/.importlinter +0 -0
  68. {setspec-0.2.0 → setspec-0.3.0}/.pre-commit-config.yaml +0 -0
  69. {setspec-0.2.0 → setspec-0.3.0}/CONTRIBUTING.md +0 -0
  70. {setspec-0.2.0 → setspec-0.3.0}/LICENSE +0 -0
  71. {setspec-0.2.0 → setspec-0.3.0}/SECURITY.md +0 -0
  72. {setspec-0.2.0 → setspec-0.3.0}/docs/packages/setspec/development-plan.md +0 -0
  73. {setspec-0.2.0 → setspec-0.3.0}/requirements/README.md +0 -0
  74. {setspec-0.2.0 → setspec-0.3.0}/requirements/release.in +0 -0
  75. {setspec-0.2.0 → setspec-0.3.0}/requirements/release.lock +0 -0
  76. {setspec-0.2.0 → setspec-0.3.0}/src/setspec/base.py +0 -0
  77. {setspec-0.2.0 → setspec-0.3.0}/src/setspec/error/v1.py +0 -0
  78. {setspec-0.2.0 → setspec-0.3.0}/src/setspec/errors.py +0 -0
  79. {setspec-0.2.0 → setspec-0.3.0}/src/setspec/event/v1.py +0 -0
  80. {setspec-0.2.0 → setspec-0.3.0}/src/setspec/metrics.py +0 -0
  81. {setspec-0.2.0 → setspec-0.3.0}/src/setspec/provenance.py +0 -0
  82. {setspec-0.2.0 → setspec-0.3.0}/src/setspec/py.typed +0 -0
  83. {setspec-0.2.0 → setspec-0.3.0}/src/setspec/serialization.py +0 -0
  84. {setspec-0.2.0 → setspec-0.3.0}/src/setspec/vocabulary.py +0 -0
  85. {setspec-0.2.0 → setspec-0.3.0}/tests/conftest.py +0 -0
  86. {setspec-0.2.0 → setspec-0.3.0}/tests/contract/test_version_negotiation.py +0 -0
  87. {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_base.py +0 -0
  88. {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_errors.py +0 -0
  89. {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_events.py +0 -0
  90. {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_metrics.py +0 -0
  91. {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_payloads_benchmark.py +0 -0
  92. {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_payloads_capability.py +0 -0
  93. {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_serialization.py +0 -0
  94. {setspec-0.2.0 → setspec-0.3.0}/tests/unit/test_vocabulary.py +0 -0
@@ -91,6 +91,44 @@ jobs:
91
91
  - run: pip install . --no-deps
92
92
  - run: pytest -m contract
93
93
 
94
+ # The three published guarantees get their own jobs rather than sharing `contracts`' single
95
+ # red mark, because each answers a different question and the answers are acted on differently:
96
+ # a snapshot diff means "bump the version", a golden failure means "the artifact is wrong", and
97
+ # a cross-version failure means "a consumer at another version just broke".
98
+ schema-snapshot-diff:
99
+ runs-on: ubuntu-latest
100
+ steps:
101
+ - uses: actions/checkout@v4
102
+ - uses: actions/setup-python@v5
103
+ with: { python-version: "3.12" }
104
+ - run: pip install --require-hashes -r requirements/ci.lock
105
+ - run: pip install . --no-deps
106
+ # ADR-0009 rule 7: a schema change without a version bump fails CI. The test regenerates
107
+ # every published schema from its writer model and diffs it against the committed snapshot.
108
+ - run: pytest tests/contract/test_schema_snapshots.py -m contract
109
+
110
+ golden-validation:
111
+ runs-on: ubuntu-latest
112
+ steps:
113
+ - uses: actions/checkout@v4
114
+ - uses: actions/setup-python@v5
115
+ with: { python-version: "3.12" }
116
+ - run: pip install --require-hashes -r requirements/ci.lock
117
+ - run: pip install . --no-deps
118
+ - run: pytest tests/contract/test_goldens.py -m contract
119
+
120
+ cross-version-compatibility:
121
+ runs-on: ubuntu-latest
122
+ steps:
123
+ - uses: actions/checkout@v4
124
+ - uses: actions/setup-python@v5
125
+ with: { python-version: "3.12" }
126
+ - run: pip install --require-hashes -r requirements/ci.lock
127
+ - run: pip install . --no-deps
128
+ # Testing Standards §8.3, proven by the repository that makes the promise: a v1.0 reader
129
+ # accepts a synthetic v1.1 golden without loss and refuses a synthetic v2.0 by major.
130
+ - run: pytest tests/contract/test_cross_version.py -m contract
131
+
94
132
  security:
95
133
  runs-on: ubuntu-latest
96
134
  steps:
@@ -148,3 +186,18 @@ jobs:
148
186
  marker = pathlib.Path(setspec.__file__).parent / 'py.typed'
149
187
  assert marker.is_file(), 'py.typed missing from the installed wheel'
150
188
  "
189
+ # The schemas and goldens are package data (ADR-0009 rule 7), and package data is exactly
190
+ # what a build is most likely to leave out silently: every test in this repository passes
191
+ # against the source tree whether or not the files reach the wheel. This is the only job
192
+ # that would notice, which is why it asserts against the *installed* distribution.
193
+ - name: schemas and goldens ship in the wheel
194
+ run: |
195
+ python -c "
196
+ from setspec import PUBLISHED_SCHEMAS, golden_payloads, json_schema_for
197
+ for schema, versions in PUBLISHED_SCHEMAS.items():
198
+ for version in versions:
199
+ assert json_schema_for(schema, version)['title'] == schema
200
+ goldens = golden_payloads(schema, version)
201
+ assert len(goldens) >= 3, f'{schema} {version} ships {len(goldens)} goldens'
202
+ print(f'{len(PUBLISHED_SCHEMAS)} schemas with goldens loaded from the installed wheel')
203
+ "
@@ -7,7 +7,84 @@ packaging and release standards §3.
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.3.0] — 2026-08-28
11
+
10
12
  ### Added
13
+
14
+ - **The v1.0 freeze.** `DRAFT_SCHEMAS` is now empty. Every registered payload type —
15
+ `model.identity`, `machine.profile`, `benchmark.result`, `benchmark.run_summary`,
16
+ `capability.evidence`, `benchmark.evidence_bundle`, `benchmark.goal_pack` and
17
+ `benchmark.calibration_report` — is a published `1.0` that may no longer be reshaped in place.
18
+ FreeWeight's real output (P6 onward, and P8A–8B for the two goal payloads) demanded no field
19
+ change, so the freeze is a promotion rather than a correction; the one field the pass did add,
20
+ `MetricValueFields.metric_key`, landed before it and is frozen *with* the schema.
21
+
22
+ The set survives its own emptiness deliberately: it is the mechanism a future draft uses, and
23
+ deleting it would leave the next provisional payload type either shipping silently provisional or
24
+ inventing a second way to say so.
25
+
26
+ - **Generated JSON Schema for every published version**, committed as package data under
27
+ `setspec/schemas/<schema>/<version>.json` and reachable through `json_schema_for()`. Generated
28
+ from the **writer** (`extra="forbid"`) half, so a published document says
29
+ `additionalProperties: false` — which is the contract API standards §7 rule 5 states and the one
30
+ a producer's test suite asserts its output against. The reader policy stays where it belongs, in
31
+ the reader: `load_envelope` accepts an unknown minor by comparing majors, never by loosening a
32
+ version's schema to admit fields it cannot describe.
33
+
34
+ Every `$ref` resolves inside the document's own `$defs`. Nothing is fetched (spec §14).
35
+
36
+ - **25 golden payloads**, three or more per version, under
37
+ `setspec/goldens/<schema>/<version>/<name>.json` and reachable through `golden_payloads()` and
38
+ `golden_names()`. Each version ships a `minimal` (only required fields), a `full` (every field
39
+ populated) and, wherever the payload carries a measurement at all, an `unsupported`-heavy example.
40
+ `capability.evidence` adds `goal`, a calibrated `user.noir_tech_voice` record with the whole
41
+ goal-sourced group populated; `benchmark.goal_pack` adds `starter_unforked`; and
42
+ `benchmark.calibration_report` adds `gate_failed` — the outcome `capability.evidence`
43
+ deliberately cannot express, because a goal below its gate emits no evidence record at all.
44
+
45
+ Which versions need an `unsupported` golden is derived from the published schema rather than
46
+ listed in the test: a payload whose document contains no `{"const": "unsupported"}` branch has no
47
+ measurement field to leave unsupported, so `benchmark.goal_pack` and
48
+ `benchmark.calibration_report` are excused by the artifact itself rather than by an exception
49
+ someone maintains.
50
+
51
+ - **`setspec.artifacts`** — `PUBLISHED_SCHEMAS`, `json_schema_for()`, `golden_payloads()`,
52
+ `golden_names()`, `payload_pair()`, `build_json_schema()` and `render_schema_document()`. The
53
+ first four are re-exported from `setspec` directly, matching spec §7's "schema artefacts" group:
54
+ an accessor that *returns* a version's artifacts is not itself versioned.
55
+
56
+ `PUBLISHED_SCHEMAS` records **exact** versions, unlike `SUPPORTED_SCHEMAS`, which records
57
+ supported majors. An artifact is a file that either exists or does not, so "highest minor known"
58
+ is the wrong shape for it — and a contract test asserts the two agree, so a schema that can be
59
+ negotiated but not validated, or validated but not negotiated, fails the build.
60
+
61
+ - **Three contract jobs, one per published guarantee.** `schema-snapshot-diff` regenerates every
62
+ schema from its models and diffs it against the committed snapshot — ADR-0009 rule 7, *a schema
63
+ change without a version bump fails CI*, made mechanical and proven by a deliberate test-only
64
+ mutation. `golden-validation` checks every golden against its writer model, its reader model, the
65
+ published JSON Schema and a canonical round trip. `cross-version-compatibility` builds a synthetic
66
+ `1.1` document from each `full` golden and asserts a `1.0` reader accepts it, preserves the field
67
+ it does not know, and re-exports it intact — then builds a synthetic `2.0` and asserts the refusal
68
+ names both versions. They are separate jobs rather than steps because the three failures are acted
69
+ on differently: bump the version, fix the artifact, or go and warn a consumer.
70
+
71
+ - **`install-check` now asserts the package data reaches the wheel.** Every test in this repository
72
+ passes against the source tree whether or not the schemas and goldens are built into the
73
+ distribution, so this job is the only one that would notice. It loads all eight schemas and their
74
+ goldens out of the *installed* wheel.
75
+
76
+ - **`docs/schemas.md`** — the human-readable catalogue: what ships and where, every payload type
77
+ with its required-field count and its goldens, how to consume the artifacts from a repository that
78
+ shares no code with this one, and a table of every cross-field rule the JSON Schema **cannot**
79
+ express and the models therefore still own.
80
+
81
+ - **`jsonschema` and `types-jsonschema` in the `dev` extra.** Test-only, and justified by the
82
+ assertion they exist for: a golden is validated against the *published document* a non-Python
83
+ consumer receives, not only against the model that generated it. Without a real validator that
84
+ check would be a hand-rolled subset of draft 2020-12 maintained in this repository — the thing
85
+ under test also serving as the thing testing it. Nothing under `src/` imports either, and the base
86
+ install still pulls in no validator. `requirements/ci.lock` regenerated.
87
+
11
88
  - **`metric_key` on `MetricValueFields`.** The model declared value, unit, aggregation, direction,
12
89
  sample count and dispersion — and nothing saying *which metric it is*. Both
13
90
  `BenchmarkResultFields.metrics` and `BenchmarkRunSummaryFields.aggregate_metrics` carry sequences
@@ -58,6 +135,14 @@ packaging and release standards §3.
58
135
 
59
136
  ### Changed
60
137
 
138
+ - The five payload modules' status notes read **frozen (`1.0`)** rather than **draft (`1.0`)**, and
139
+ say what the freeze binds them to: a new optional field is a minor bump, and a removal, rename,
140
+ retype or tightening is a major — never an edit in place.
141
+ - The two tests that asserted the draft state now assert the frozen one
142
+ (`test_no_registered_schema_is_still_draft`, `test_the_schema_is_frozen`). They are the same
143
+ guarantee read from the other side, and leaving them asserting `DRAFT_SCHEMAS == SUPPORTED_SCHEMAS`
144
+ would have made the freeze a change the test suite refused.
145
+
61
146
  - A bare reserved root is now refused by `validate_capability` and reported `False` by
62
147
  `is_known_capability`. `user` is a namespace, not a capability: a payload claiming
63
148
  `capability_id: "user"` has lost the identity that is the entire point of the namespace.
@@ -71,6 +156,7 @@ packaging and release standards §3.
71
156
  declaring `1.2` or later is unaffected. Producers on `1.1` must emit roots that exist at `1.1`.
72
157
 
73
158
  ### Fixed
159
+
74
160
  - **Coverage measured a directory nothing imports.** `[tool.coverage.run] source` named
75
161
  `src/setspec`, the source *path*, while CI installs the built distribution — so the moment the
76
162
  jobs stopped using an editable install, coverage reported **0 %** and failed the 95 % floor for a
@@ -89,6 +175,13 @@ packaging and release standards §3.
89
175
  (`baseaicore`, `modelrack`, `sweatmeter`). Without it, `mypy --strict` in a consuming repository
90
176
  cannot see this package's types at all and treats every import from it as untyped.
91
177
 
178
+ ### Known gaps
179
+
180
+ - `event.envelope` and `error.envelope` (Phase 3) and `prompt.record` / `prompt.manifest`
181
+ (Phase 5) are **not** part of this freeze, because they do not exist yet: a schema is frozen by
182
+ being published, and an unwritten one has nothing to publish. The freeze covers the eight payload
183
+ types that cross an application boundary today, not the eleven ADR-0009 anticipates.
184
+
92
185
  ## [0.2.0] — 2026-08-23
93
186
 
94
187
  Phase 2 of the [development plan](docs/packages/setspec/development-plan.md): provisional
@@ -0,0 +1,99 @@
1
+ # Phase 4 — issues to address
2
+
3
+ Written at the end of Phase 4 (freeze v1.0, publish schemas and goldens; `setspec 0.3.0`). Each
4
+ entry is something a later phase, a docs change, a consumer or a release step has to resolve.
5
+ Nothing here blocks Phase 4's acceptance criteria; everything here would become a defect if it were
6
+ forgotten.
7
+
8
+ ---
9
+
10
+ ## Status — 2026-08-28
11
+
12
+ | # | Issue | Status |
13
+ |---|---|---|
14
+ | 1 | The freeze covers eight payload types, not ADR-0009's eleven | **Open — by design.** Phases 3 and 5 own the other three. |
15
+ | 2 | Goldens are authored inputs, but a SetSpec writer emits every declared key | **Needs a decision.** |
16
+ | 3 | `ContributingMetricFields.metric_key` carries no pattern while `MetricValueFields.metric_key` does | **Needs a decision.** |
17
+ | 4 | Cross-field rules are invisible in the published JSON Schema | **Open — documented.** `docs/schemas.md` §4 lists every one. |
18
+ | 5 | `jsonschema` joined the `dev` extra; `requirements/ci.lock` regenerated | **Closed** — verify `pip-audit` in CI. |
19
+ | 6 | The release is not tagged or published yet | **Owner: you.** Commands below. |
20
+
21
+ ---
22
+
23
+ ## 1. The freeze covers eight payload types, not eleven
24
+
25
+ ADR-0009 lists eleven initial payload types. `event.envelope` and `error.envelope` (Phase 3) and
26
+ `prompt.record` / `prompt.manifest` (Phase 5) are not written, so they are not frozen — a schema is
27
+ frozen by being published, and an unwritten one has nothing to publish. `DRAFT_SCHEMAS` is empty
28
+ **now**; when Phase 3 or 5 lands a new payload type it should re-enter that set until its own
29
+ freeze, which is exactly the mechanism the set survives its emptiness for. The changelog's *Known
30
+ gaps* section says the same. Nothing to do until those phases start, except not to read "empty"
31
+ as "finished".
32
+
33
+ ## 2. Goldens are authored inputs, but a SetSpec writer emits every declared key
34
+
35
+ `dump_envelope(CapabilityEvidenceOut(...))` serialises through `model_dump()`, which writes every
36
+ declared field — a non-goal record carries `goal_hash: null`, `uncalibrated: false`, and so on. The
37
+ `capability.evidence/1.0/full` golden was authored the other way round: it omits the goal group
38
+ entirely, because "fully populated" was read as "every field a *non-goal* record populates". Both
39
+ forms validate under both models, so the contract is not broken, but a producer's structural test
40
+ ("my keys match the full golden's keys") fails against the file and passes against the file *as a
41
+ SetSpec writer would dump it*. FreeWeight's contract test compares against the latter.
42
+
43
+ **Decision needed:** should goldens be committed *as written* (every declared key present, nulls
44
+ included), so that "matches the golden structurally" is a byte-level statement? If yes, regenerate
45
+ the 25 goldens through their `Out` models once and add a contract test that a golden equals its own
46
+ writer-dumped form. If no, say so in `docs/schemas.md` §3 so consumers compare against the dumped
47
+ form, as FreeWeight now does.
48
+
49
+ ## 3. `ContributingMetricFields.metric_key` has no pattern
50
+
51
+ `MetricValueFields.metric_key` is constrained to lower snake case, dot-separable
52
+ (`^[a-z][a-z0-9_]*(\.[a-z][a-z0-9_]*)*$`). `ContributingMetricFields.metric_key` — the key inside
53
+ `capability.evidence.contributing_metrics` — is only `min_length=1`. FreeWeight writes
54
+ `<suite_key>.<metric_key>` there (`native.tool_use.task_success`), `criterion.<key>` for a goal's
55
+ own record and `goal.<slug>.composite_score` for a goal contributing to a shipped capability; all
56
+ three happen to satisfy the stricter pattern. Tightening the field is a **major** change now that
57
+ `1.0` is frozen, so it can only land as `capability.evidence 2.0`, and only if a consumer ever
58
+ needs the guarantee. Recorded so the asymmetry is a decision rather than an accident.
59
+
60
+ ## 4. Cross-field rules are invisible in the published JSON Schema
61
+
62
+ Pydantic renders types, ranges, patterns and required keys; it cannot render a `model_validator`.
63
+ `runtime_profile_hash` agreeing with its profile, `score_method_mix` summing to one,
64
+ `measured_at ≤ computed_at`, the goal group's five coherence rules — every one is enforced by the
65
+ models only. `docs/schemas.md` §4 tables them, and every golden is validated against *both* the
66
+ schema and the model for exactly this reason. A non-Python consumer that needs those rules needs a
67
+ second validation step this package does not ship. Nothing to do unless such a consumer appears.
68
+
69
+ ## 5. `jsonschema` in the `dev` extra
70
+
71
+ Added so the golden contract test validates against the *published* document with a real
72
+ draft-2020-12 validator rather than a hand-rolled subset. Test only — nothing under `src/`
73
+ imports it and the base install pulls in no validator. `requirements/ci.lock` was regenerated with
74
+ `pip-compile --generate-hashes` (jsonschema 4.26.0, jsonschema-specifications, referencing,
75
+ rpds-py, attrs, types-jsonschema). `release.lock` is unchanged. CI's `security` job audits both
76
+ locks; the first run after this lands is the one to watch.
77
+
78
+ ## 6. Tagging and publishing 0.3.0
79
+
80
+ Nothing was tagged or published. `__about__.py` says `0.3.0` and the changelog carries a dated
81
+ `[0.3.0]` section. The release procedure (packaging standards §6), from the SetSpec repository:
82
+
83
+ ```bash
84
+ cd ~/ai/suite/py/SetSpec
85
+ git add -A
86
+ git commit -m "feat(setspec): freeze v1.0, publish JSON Schema and goldens (Phase 4, 0.3.0)"
87
+ git push origin main
88
+ # wait for CI to be green on main, then:
89
+ git tag -a v0.3.0 -m "setspec 0.3.0 — v1.0 contracts frozen; JSON Schema and goldens published"
90
+ git push origin v0.3.0
91
+ # release.yml builds, tests the wheel, publishes via Trusted Publishing and creates the release.
92
+ # Verify:
93
+ python -m venv /tmp/setspec-check && /tmp/setspec-check/bin/pip install setspec==0.3.0 && \
94
+ /tmp/setspec-check/bin/python -c "from setspec import PUBLISHED_SCHEMAS, golden_payloads, SchemaVersion; \
95
+ print(len(PUBLISHED_SCHEMAS), len(golden_payloads('capability.evidence', SchemaVersion(1, 0))))"
96
+ ```
97
+
98
+ FreeWeight's `pyproject.toml` already pins `setspec>=0.3,<0.4`; its `install-check` job cannot
99
+ resolve until this release is on PyPI.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: setspec
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: Every versioned data contract that crosses an application boundary: benchmark results, capability evidence, event/error envelopes, prompt records.
5
5
  Project-URL: Homepage, https://github.com/JPKell/SetSpec
6
6
  Project-URL: Documentation, https://github.com/JPKell/SetSpec/tree/main/docs
@@ -21,25 +21,31 @@ Requires-Dist: jinja2<4,>=3.1
21
21
  Requires-Dist: pydantic<3,>=2.9
22
22
  Provides-Extra: dev
23
23
  Requires-Dist: import-linter<3,>=2.0; extra == 'dev'
24
+ Requires-Dist: jsonschema<5,>=4.23; extra == 'dev'
24
25
  Requires-Dist: mypy<2,>=1.11; extra == 'dev'
25
26
  Requires-Dist: pytest-cov<6,>=5; extra == 'dev'
26
27
  Requires-Dist: pytest-randomly<4,>=3; extra == 'dev'
27
28
  Requires-Dist: pytest<10,>=9.0.3; extra == 'dev'
28
29
  Requires-Dist: respx<1,>=0.21; extra == 'dev'
29
30
  Requires-Dist: ruff<1,>=0.6; extra == 'dev'
31
+ Requires-Dist: types-jsonschema<5,>=4.23; extra == 'dev'
30
32
  Description-Content-Type: text/markdown
31
33
 
32
34
  # SetSpec
33
35
 
34
36
  Every versioned data contract that crosses an application boundary: benchmark results, capability evidence, event/error envelopes, prompt records.
35
37
 
36
- **Status:** `0.2.0` — Phases 1–2 complete. The envelope, version negotiation and serialization core
37
- are implemented and tested; `model.identity`, `machine.profile`, `benchmark.result`,
38
- `benchmark.run_summary`, `capability.evidence` and `benchmark.evidence_bundle` are registered in
39
- `SUPPORTED_SCHEMAS`, but **draft**: Phase 4 may still reshape a field once FreeWeight has produced
40
- real results against these payloads. Which schemas are still provisional is readable at runtime
41
- from `setspec.DRAFT_SCHEMAS`, not only stated here — freezing one is a deletion from that set.
42
- Event and error envelopes arrive in Phase 3. See the
38
+ **Status:** `0.3.0` — Phases 1–2, 3A and 4 complete, and **the v1.0 contracts are frozen.** Eight
39
+ payload types are published at `1.0` — `model.identity`, `machine.profile`, `benchmark.result`,
40
+ `benchmark.run_summary`, `capability.evidence`, `benchmark.evidence_bundle`,
41
+ `benchmark.goal_pack` and `benchmark.calibration_report` — each with generated JSON Schema and at
42
+ least three golden payloads shipped as package data. `setspec.DRAFT_SCHEMAS` is empty, which is
43
+ where the freeze is readable at runtime rather than only stated here; from now on a new optional
44
+ field is a minor bump and anything else is a major, enforced by a snapshot diff in CI.
45
+
46
+ The [schema catalogue](docs/schemas.md) lists every payload type, its artifacts, and the
47
+ cross-field rules the JSON Schema cannot express. Event and error envelopes (Phase 3) and prompt
48
+ records (Phase 5) are not yet written and are therefore not part of the freeze. See the
43
49
  [development plan](docs/packages/setspec/development-plan.md) for what each phase adds.
44
50
 
45
51
  Part of the **Local AI Suite**.
@@ -2,13 +2,17 @@
2
2
 
3
3
  Every versioned data contract that crosses an application boundary: benchmark results, capability evidence, event/error envelopes, prompt records.
4
4
 
5
- **Status:** `0.2.0` — Phases 1–2 complete. The envelope, version negotiation and serialization core
6
- are implemented and tested; `model.identity`, `machine.profile`, `benchmark.result`,
7
- `benchmark.run_summary`, `capability.evidence` and `benchmark.evidence_bundle` are registered in
8
- `SUPPORTED_SCHEMAS`, but **draft**: Phase 4 may still reshape a field once FreeWeight has produced
9
- real results against these payloads. Which schemas are still provisional is readable at runtime
10
- from `setspec.DRAFT_SCHEMAS`, not only stated here — freezing one is a deletion from that set.
11
- Event and error envelopes arrive in Phase 3. See the
5
+ **Status:** `0.3.0` — Phases 1–2, 3A and 4 complete, and **the v1.0 contracts are frozen.** Eight
6
+ payload types are published at `1.0` — `model.identity`, `machine.profile`, `benchmark.result`,
7
+ `benchmark.run_summary`, `capability.evidence`, `benchmark.evidence_bundle`,
8
+ `benchmark.goal_pack` and `benchmark.calibration_report` — each with generated JSON Schema and at
9
+ least three golden payloads shipped as package data. `setspec.DRAFT_SCHEMAS` is empty, which is
10
+ where the freeze is readable at runtime rather than only stated here; from now on a new optional
11
+ field is a minor bump and anything else is a major, enforced by a snapshot diff in CI.
12
+
13
+ The [schema catalogue](docs/schemas.md) lists every payload type, its artifacts, and the
14
+ cross-field rules the JSON Schema cannot express. Event and error envelopes (Phase 3) and prompt
15
+ records (Phase 5) are not yet written and are therefore not part of the freeze. See the
12
16
  [development plan](docs/packages/setspec/development-plan.md) for what each phase adds.
13
17
 
14
18
  Part of the **Local AI Suite**.
@@ -4,5 +4,6 @@ This directory contains the maintained documentation for SetSpec.
4
4
 
5
5
  ## Documents
6
6
 
7
+ - [Schema catalogue](schemas.md) — every payload type and version, with its artifacts
7
8
  - [Development plan](packages/setspec/development-plan.md)
8
9
  - [Specification](packages/setspec/spec.md)
@@ -124,9 +124,11 @@ RESERVED_ROOTS: frozenset[str] # roots valid ONLY as a specialization; {"
124
124
  validate_capability(capability_id: str) -> CapabilityId
125
125
  is_known_capability(capability_id: str) -> bool
126
126
 
127
- # Schema artefacts
128
- json_schema_for(schema: str, version: SchemaVersion) -> dict
129
- golden_payloads(schema: str, version: SchemaVersion) -> list[dict]
127
+ # Schema artefacts (package data; every published version keeps both for the life of its major)
128
+ PUBLISHED_SCHEMAS: Mapping[str, tuple[SchemaVersion, ...]] # exact versions with committed artefacts
129
+ json_schema_for(schema: str, version: SchemaVersion) -> dict # the committed JSON Schema snapshot
130
+ golden_payloads(schema: str, version: SchemaVersion) -> list[dict] # ≥ 3 per version, by name
131
+ golden_names(schema: str, version: SchemaVersion) -> tuple[str, ...] # the same order as above
130
132
 
131
133
  # Prompts (ADR-0028) — one implementation of a determinism contract, not three
132
134
  class PromptLibrary:
@@ -0,0 +1,195 @@
1
+ # Schema catalogue
2
+
3
+ Every payload type `setspec` publishes, at every version, with the artifacts that make it usable
4
+ from a repository that shares no code with this one.
5
+
6
+ **Status: frozen at `1.0`** (Phase 4, `setspec 0.3.0`). `DRAFT_SCHEMAS` is empty. From here every
7
+ change follows the ordinary rules: a new optional field is a **minor** bump, and a removed,
8
+ renamed, retyped or newly-tightened field is a **major**. Neither happens by editing a payload
9
+ module in place — the snapshot contract test fails the build if the generated schema stops matching
10
+ the committed one, which is ADR-0009 rule 7 made mechanical.
11
+
12
+ ---
13
+
14
+ ## 1. What ships, and where it lives
15
+
16
+ | Artifact | Location in the installed package | Accessor |
17
+ |---|---|---|
18
+ | JSON Schema, one per version | `setspec/schemas/<schema>/<version>.json` | `json_schema_for(schema, version)` |
19
+ | Golden payloads, ≥ 3 per version | `setspec/goldens/<schema>/<version>/<name>.json` | `golden_payloads(schema, version)` |
20
+ | The names of those goldens | — | `golden_names(schema, version)` |
21
+ | Every published schema and version | — | `PUBLISHED_SCHEMAS` |
22
+
23
+ All four are importable from `setspec` directly. The files are **package data**, loadable through
24
+ `importlib.resources` and present in the built wheel — the `install-check` CI job asserts that
25
+ against the installed distribution, because every test in this repository passes against the source
26
+ tree whether or not the files reach the wheel.
27
+
28
+ Nothing is fetched. A published schema resolves every `$ref` inside its own `$defs`, so validating
29
+ an export never depends on a registry being reachable (spec §14).
30
+
31
+ ## 2. The catalogue
32
+
33
+ | Schema | Version | Python module | Writer / reader | Goldens |
34
+ |---|---|---|---|---|
35
+ | `model.identity` | 1.0 | `setspec.model.v1` | `ModelIdentityOut` / `ModelIdentityIn` | `minimal`, `full`, `unsupported` |
36
+ | `machine.profile` | 1.0 | `setspec.machine.v1` | `MachineProfileOut` / `MachineProfileIn` | `minimal`, `full`, `unsupported` |
37
+ | `benchmark.result` | 1.0 | `setspec.benchmark.v1` | `BenchmarkResultOut` / `BenchmarkResultIn` | `minimal`, `full`, `unsupported` |
38
+ | `benchmark.run_summary` | 1.0 | `setspec.benchmark.v1` | `BenchmarkRunSummaryOut` / `BenchmarkRunSummaryIn` | `minimal`, `full`, `unsupported` |
39
+ | `capability.evidence` | 1.0 | `setspec.capability.v1` | `CapabilityEvidenceOut` / `CapabilityEvidenceIn` | `minimal`, `full`, `goal`, `unsupported` |
40
+ | `benchmark.evidence_bundle` | 1.0 | `setspec.capability.v1` | `EvidenceBundleOut` / `EvidenceBundleIn` | `minimal`, `full`, `unsupported` |
41
+ | `benchmark.goal_pack` | 1.0 | `setspec.goal.v1` | `GoalPackOut` / `GoalPackIn` | `minimal`, `full`, `starter_unforked` |
42
+ | `benchmark.calibration_report` | 1.0 | `setspec.goal.v1` | `CalibrationReportOut` / `CalibrationReportIn` | `minimal`, `full`, `gate_failed` |
43
+
44
+ ### `model.identity` — which weights, plus what the provider says about them
45
+
46
+ 5 required fields of 25. The identity triple (`provider_kind`, `provider_model_name`,
47
+ `artifact_digest`) plus its two derived fields, then the refreshable descriptor. `canonical_id` and
48
+ `identity_confidence` are **recomputed and checked** against the triple on validation: they are
49
+ pure functions of it, carried on the wire for convenience rather than as independent facts
50
+ (ADR-0024). Every descriptor quantity defaults to `"unsupported"`, because a provider that reports
51
+ no layer count has not reported zero layers.
52
+
53
+ ### `machine.profile` — where a measurement happened
54
+
55
+ 7 required of 14. A field-for-field mirror of `baseaicore.MachineProfile`, including which fields
56
+ are required-but-nullable: a producer must say it could not read the hostname rather than omit the
57
+ key. `machine_fingerprint` is the one hash-shaped field this package carries **without**
58
+ recomputing it — the policy deciding which fields feed a fingerprint may change, while a profile
59
+ read back years later must still reconstruct exactly as stored.
60
+
61
+ ### `benchmark.result` — one benchmark, one subject, one machine
62
+
63
+ 12 required of 23, the largest payload in the suite. Carries Machine Identity §6's minimum
64
+ provenance set as nested objects: `suite`, `execution`, `environment`, `application`,
65
+ `reproducibility`, plus the optional `machine_profile` and `telemetry_summary`.
66
+ `runtime_profile_hash` is recomputed from the embedded `runtime_profile` and must agree
67
+ (ADR-0023). Four further cross-field rules apply — see §4.
68
+
69
+ ### `benchmark.run_summary` — the roll-up of many results
70
+
71
+ 10 required of 15. Deliberately lighter than a result: the same subject/suite/environment building
72
+ blocks, a run state, three timestamps and `aggregate_metrics`. `INTERRUPTED` is a distinct status
73
+ from `FAILED` and means the process died and the run is resumable, not that it failed.
74
+
75
+ ### `capability.evidence` — the record LoadCoach routes on
76
+
77
+ 14 required of 26, and the suite's most load-bearing contract (ADR-0022). Identity, score,
78
+ confidence, the counts behind them, the hard-separation inputs (`benchmark_versions`,
79
+ `dataset_hashes`, `prompt_subset_hashes`, `goal_hash`, `judge_set`) and the two timestamps whose
80
+ order is enforced: `measured_at` is what freshness decays from and can never follow `computed_at`.
81
+
82
+ The **goal-sourced group** (ADR-0032 §5) is optional and absent on an ordinary record:
83
+ `judge_validity_factor` (exactly `1.0` for every rung 1–4 measurement), `goal_hash`,
84
+ `goal_pack_version`, `score_method_mix`, `judge_set`, `calibration`, `uncalibrated`. The `goal`
85
+ golden populates all of it; `uncalibrated: true` is **refused** rather than published, because a
86
+ goal below its calibration gate emits no record at all.
87
+
88
+ ### `benchmark.evidence_bundle` — the FreeWeight → LoadCoach payload
89
+
90
+ 2 required of 3: `source_id`, `complete`, `evidence`. `complete: true` is what lets a consumer
91
+ infer removal — evidence held locally for this `source_id` and absent from a complete bundle is
92
+ marked superseded, never deleted, and never inferred from a partial bundle. The bundle carries no
93
+ `generated_at`: that lives on the envelope, and a client stores *that* value to send back as its
94
+ next `?since=`.
95
+
96
+ ### `benchmark.goal_pack` — a user-authored goal, portable and hash-pinned
97
+
98
+ 6 required of 12. Criteria with their ladder rung and weight, tasks with their prompt hashes, and
99
+ the jury when any criterion is judged. Weights must sum to `1`; a judged criterion needs an
100
+ anchored ordinal scale and a `judge_set`. The author's calibration **grades do not travel** — an
101
+ importer who wants the rubric held to their own taste recalibrates against their own grades.
102
+
103
+ ### `benchmark.calibration_report` — how well the judge agreed with the author
104
+
105
+ 11 required of 14. A payload in its own right rather than a field group on evidence, because it is
106
+ meaningful precisely when **no** evidence was emitted: `gate_failed` is the golden that shows the
107
+ outcome `capability.evidence` deliberately cannot express (ADR-0032 §3). The verdict must follow
108
+ from the numbers — a `passed_gate` that contradicts `weighted_kappa_w` against `min_agreement` is
109
+ refused.
110
+
111
+ ## 3. Consuming these from another repository
112
+
113
+ No shared code, no import of the producing application, no database:
114
+
115
+ ```python
116
+ from setspec import SchemaVersion, golden_payloads, json_schema_for, load_envelope
117
+ from setspec.capability.v1 import EvidenceBundleIn
118
+
119
+ # The contract test: read every golden this build publishes.
120
+ for payload in golden_payloads("benchmark.evidence_bundle", SchemaVersion(1, 0)):
121
+ EvidenceBundleIn.model_validate(payload)
122
+
123
+ # The real thing: a file a producer exported.
124
+ envelope = load_envelope(exported_bytes, expect="benchmark.evidence_bundle")
125
+ bundle = EvidenceBundleIn.model_validate(envelope.payload)
126
+ ```
127
+
128
+ Read with the **`In`** model, always. It preserves fields this build has not heard of, so an older
129
+ consumer in the middle of a pipeline is a relay rather than a sink (ADR-0009 rule 4). Write with the
130
+ **`Out`** model, which refuses a field the schema does not declare (rule 5).
131
+
132
+ A non-Python consumer validates against `json_schema_for(...)`, or against the file directly:
133
+
134
+ ```
135
+ setspec/schemas/benchmark.evidence_bundle/1.0.json
136
+ ```
137
+
138
+ The documents declare `$schema: https://json-schema.org/draft/2020-12/schema`.
139
+
140
+ ## 4. What the JSON Schema does not say
141
+
142
+ The published documents are generated from the **writer** model, so they carry
143
+ `additionalProperties: false` at the top level — that is the writer's contract, and a producer's
144
+ test suite asserts its output validates against it (testing standards §8.2). The reader policy is a
145
+ *reader behaviour*: `load_envelope` accepts any minor within a supported major, including one it has
146
+ never heard of. That is not a loosening of any one version's schema. A `1.1` document is described
147
+ by the `1.1` schema; a `1.0` reader accepts it by comparing majors.
148
+
149
+ Pydantic renders types, ranges, patterns and required keys. It cannot render a `model_validator`, so
150
+ none of the following appears in the published documents and all of them are enforced by the models:
151
+
152
+ | Payload | Rule the schema cannot state |
153
+ |---|---|
154
+ | `metric.value` (nested) | A real value needs ≥ 1 supported sample; an unsupported value needs 0; dispersion needs ≥ 2 |
155
+ | `model.identity` | `canonical_id` and `identity_confidence` must agree with the identity triple |
156
+ | `benchmark.result` | `runtime_profile_hash` must recompute from `runtime_profile`; `completed_at ≥ started_at`; `completed_cases ≤ total_cases`; `skip_reason` iff skipped; a completed result has metrics |
157
+ | `benchmark.run_summary` | `runtime_profile_hash` agreement; timing order |
158
+ | `capability.evidence` | `capability_id` in the vocabulary; `measured_at ≤ computed_at`; the five goal-group coherence rules; `score_method_mix` sums to 1 over known rungs |
159
+ | `benchmark.goal_pack` | Criterion weights sum to 1; no duplicate keys; rung-appropriate fields; a judged criterion needs a jury |
160
+ | `benchmark.calibration_report` | The gate verdict must follow from `weighted_kappa_w` against `min_agreement` |
161
+
162
+ This is why every golden is validated **twice** — once against the schema a non-Python consumer
163
+ uses, once against the model — and why a consumer that can run Python should use the model.
164
+
165
+ ## 5. Versioning and regeneration
166
+
167
+ A schema version is `MAJOR.MINOR`, independent of this package's own version and of any HTTP API
168
+ version. `SUPPORTED_SCHEMAS` declares supported **majors**; `PUBLISHED_SCHEMAS` declares the exact
169
+ versions that have artifacts. A contract test asserts the two agree, so a schema that can be
170
+ negotiated but not validated — or validated but not negotiated — fails the build.
171
+
172
+ Adding a version means adding a module (`setspec.benchmark.v2`), registering it, and committing its
173
+ schema and goldens. The old version stays importable for at least one minor release of every
174
+ consumer (ADR-0009 rule 6).
175
+
176
+ Regenerate the snapshots after any deliberate model change:
177
+
178
+ ```bash
179
+ python - <<'PY'
180
+ from pathlib import Path
181
+ from setspec.artifacts import PUBLISHED_SCHEMAS, build_json_schema, render_schema_document
182
+
183
+ for schema, versions in PUBLISHED_SCHEMAS.items():
184
+ for version in versions:
185
+ path = Path("src/setspec/schemas", schema, f"{version}.json")
186
+ path.parent.mkdir(parents=True, exist_ok=True)
187
+ path.write_text(
188
+ render_schema_document(build_json_schema(schema, version)), encoding="utf-8"
189
+ )
190
+ PY
191
+ ```
192
+
193
+ Then answer the question CI just asked: was that change additive, or did it deserve a major bump?
194
+ Goldens are **not** regenerated. They are authored, committed and left alone — a golden that is
195
+ rebuilt from the model whenever the model changes is not a golden, it is a mirror.
@@ -36,6 +36,13 @@ dev = [
36
36
  "ruff>=0.6,<1",
37
37
  "import-linter>=2.0,<3",
38
38
  "respx>=0.21,<1",
39
+ # Phase 4's contract tests validate every golden against the *published* JSON Schema, not only
40
+ # against the model that generated it. Without a real validator that assertion would be a
41
+ # hand-rolled subset of draft 2020-12 written in this repository — which is to say, the thing
42
+ # under test would also be the thing testing it. Test-only: nothing under `src/` imports it,
43
+ # and `setspec` itself pulls in no validator (packaging standards §4).
44
+ "jsonschema>=4.23,<5",
45
+ "types-jsonschema>=4.23,<5",
39
46
  ]
40
47
 
41
48
  [project.urls]