driftproof 0.8.1 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +11 -5
- package/bin/driftproof +171 -22
- package/config.js +26 -2
- package/lib/checks.js +28 -6
- package/lib/diff.js +58 -4
- package/lib/importers.js +8 -3
- package/lib/judge.js +69 -12
- package/lib/models.js +52 -11
- package/lib/provider.js +35 -11
- package/lib/receipt.js +51 -11
- package/lib/run.js +221 -20
- package/lib/skill.js +26 -4
- package/lib/stats.js +12 -3
- package/lib/usage.js +56 -4
- package/lib/value.js +4 -2
- package/lib/verdict.js +6 -1
- package/package.json +3 -2
- package/spec/RECEIPT.md +68 -11
- package/spec/receipt.schema.json +299 -54
- package/spec/receipt.v0.5.schema.json +1318 -0
|
@@ -0,0 +1,1318 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://driftproofhq.com/spec/receipt.v0.5.schema.json",
|
|
4
|
+
"title": "driftproof receipt",
|
|
5
|
+
"description": "A hash-verified, dated record of running one agent skill eval suite with and without the skill on one model version, with the GENERATION sampled n times per arm and the judge sampled k times inside each draw. Receipt spec v0.5 — additive over v0.4: adds per-case `generation` carrying the full draw list (each draw with its own generation_hash, nested judge samples, mean and stddev), the across-draw mean and sd, the mean judge-level sd, and the variance_ratio between them (null when the judge sd is zero, never a division result); adds the sampling policy actually applied (n_planned, n_drawn, n_measured, n_unmeasured, stopping_reason); records a timed-out draw as status `unmeasured` with NO score, excluded from every statistic rather than counted as zero; and adds a per-suite canary. All v0.4 semantics are unchanged and every prior receipt still validates against its own frozen schema (v0.1, v0.2, v0.3, v0.3.1, v0.4).",
|
|
6
|
+
"type": "object",
|
|
7
|
+
"additionalProperties": false,
|
|
8
|
+
"required": [
|
|
9
|
+
"schema_version",
|
|
10
|
+
"skill",
|
|
11
|
+
"suite",
|
|
12
|
+
"run",
|
|
13
|
+
"results",
|
|
14
|
+
"comparison",
|
|
15
|
+
"verification_level",
|
|
16
|
+
"receipt_hash"
|
|
17
|
+
],
|
|
18
|
+
"allOf": [
|
|
19
|
+
{
|
|
20
|
+
"description": "TESTED tightening — the interop relaxations (null hashes, external surface, null comparison, transcripts 'none') are ONLY available to receipts below TESTED. A TESTED receipt must carry the full evidence chain, exactly as before the interop revision.",
|
|
21
|
+
"if": {
|
|
22
|
+
"required": [
|
|
23
|
+
"verification_level"
|
|
24
|
+
],
|
|
25
|
+
"properties": {
|
|
26
|
+
"verification_level": {
|
|
27
|
+
"const": "TESTED"
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
},
|
|
31
|
+
"then": {
|
|
32
|
+
"properties": {
|
|
33
|
+
"skill": {
|
|
34
|
+
"properties": {
|
|
35
|
+
"content_hash": {
|
|
36
|
+
"type": "string"
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
},
|
|
40
|
+
"suite": {
|
|
41
|
+
"properties": {
|
|
42
|
+
"format": {
|
|
43
|
+
"const": "agentskills.io/evals"
|
|
44
|
+
},
|
|
45
|
+
"suite_hash": {
|
|
46
|
+
"type": "string"
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
},
|
|
50
|
+
"run": {
|
|
51
|
+
"properties": {
|
|
52
|
+
"surface": {
|
|
53
|
+
"enum": [
|
|
54
|
+
"api",
|
|
55
|
+
"claude-cli",
|
|
56
|
+
"openai-api",
|
|
57
|
+
"openai-cli"
|
|
58
|
+
]
|
|
59
|
+
},
|
|
60
|
+
"transcripts": {
|
|
61
|
+
"enum": [
|
|
62
|
+
"retained-local",
|
|
63
|
+
"hashes-only"
|
|
64
|
+
]
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
},
|
|
68
|
+
"comparison": {
|
|
69
|
+
"properties": {
|
|
70
|
+
"baseline_score": {
|
|
71
|
+
"type": "number"
|
|
72
|
+
},
|
|
73
|
+
"delta": {
|
|
74
|
+
"type": "number"
|
|
75
|
+
},
|
|
76
|
+
"delta_uncertainty": {
|
|
77
|
+
"type": "number"
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
},
|
|
81
|
+
"results": {
|
|
82
|
+
"properties": {
|
|
83
|
+
"cases": {
|
|
84
|
+
"items": {
|
|
85
|
+
"if": {
|
|
86
|
+
"not": {
|
|
87
|
+
"required": [
|
|
88
|
+
"case_status"
|
|
89
|
+
],
|
|
90
|
+
"properties": {
|
|
91
|
+
"case_status": {
|
|
92
|
+
"const": "failed_timeout"
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
},
|
|
97
|
+
"then": {
|
|
98
|
+
"required": [
|
|
99
|
+
"generation_hash",
|
|
100
|
+
"judge_sample_hashes"
|
|
101
|
+
],
|
|
102
|
+
"properties": {
|
|
103
|
+
"judge": {
|
|
104
|
+
"properties": {
|
|
105
|
+
"rubric_hash": {
|
|
106
|
+
"type": "string"
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
},
|
|
119
|
+
{
|
|
120
|
+
"$comment": "driftproof/generation-sampled-declared",
|
|
121
|
+
"description": "F-014-F, first direction: a receipt carrying a draw set or a variance ratio must declare the capability. Without this a third-party emitter validates as v0.5 while carrying none of what v0.5 exists to add.",
|
|
122
|
+
"if": {
|
|
123
|
+
"required": [
|
|
124
|
+
"results"
|
|
125
|
+
],
|
|
126
|
+
"properties": {
|
|
127
|
+
"results": {
|
|
128
|
+
"required": [
|
|
129
|
+
"cases"
|
|
130
|
+
],
|
|
131
|
+
"properties": {
|
|
132
|
+
"cases": {
|
|
133
|
+
"contains": {
|
|
134
|
+
"required": [
|
|
135
|
+
"generation"
|
|
136
|
+
]
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
},
|
|
143
|
+
"then": {
|
|
144
|
+
"required": [
|
|
145
|
+
"generation_sampled"
|
|
146
|
+
],
|
|
147
|
+
"properties": {
|
|
148
|
+
"generation_sampled": {
|
|
149
|
+
"const": true
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
},
|
|
154
|
+
{
|
|
155
|
+
"$comment": "driftproof/generation-sampled-honest",
|
|
156
|
+
"description": "F-014-F, second direction: a receipt that declares the capability must carry it. A flag assertable by a receipt carrying nothing would be a second way to claim a capability falsely, and would close the exposure in one direction only.",
|
|
157
|
+
"if": {
|
|
158
|
+
"required": [
|
|
159
|
+
"generation_sampled"
|
|
160
|
+
],
|
|
161
|
+
"properties": {
|
|
162
|
+
"generation_sampled": {
|
|
163
|
+
"const": true
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
},
|
|
167
|
+
"then": {
|
|
168
|
+
"required": [
|
|
169
|
+
"results"
|
|
170
|
+
],
|
|
171
|
+
"properties": {
|
|
172
|
+
"results": {
|
|
173
|
+
"required": [
|
|
174
|
+
"cases"
|
|
175
|
+
],
|
|
176
|
+
"properties": {
|
|
177
|
+
"cases": {
|
|
178
|
+
"contains": {
|
|
179
|
+
"required": [
|
|
180
|
+
"generation"
|
|
181
|
+
]
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
},
|
|
189
|
+
{
|
|
190
|
+
"$comment": "driftproof/generation-sampled-canary",
|
|
191
|
+
"description": "F-015-A, the other half of F-014-F: a receipt declaring the capability must also carry the per-suite canary. F-014-F named BOTH results.cases[].generation and suite.canary; spec 015 bound the first and left this one, so a receipt could still claim v0.5 conformance while carrying only half of what v0.5 adds. A receipt that declares generation_sampled has by construction run a suite of ours, so it has a canary to record; omitting it is the same false claim in the other half. Legacy is untouched: a v0.4-and-earlier receipt is governed by its own frozen schema, and a v0.5 receipt that ran no generation sampling declares nothing and is unaffected.",
|
|
192
|
+
"if": {
|
|
193
|
+
"required": [
|
|
194
|
+
"generation_sampled"
|
|
195
|
+
],
|
|
196
|
+
"properties": {
|
|
197
|
+
"generation_sampled": {
|
|
198
|
+
"const": true
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
},
|
|
202
|
+
"then": {
|
|
203
|
+
"required": [
|
|
204
|
+
"suite"
|
|
205
|
+
],
|
|
206
|
+
"properties": {
|
|
207
|
+
"suite": {
|
|
208
|
+
"required": [
|
|
209
|
+
"canary"
|
|
210
|
+
]
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
],
|
|
216
|
+
"properties": {
|
|
217
|
+
"schema_version": {
|
|
218
|
+
"type": "string",
|
|
219
|
+
"const": "0.5"
|
|
220
|
+
},
|
|
221
|
+
"skill": {
|
|
222
|
+
"type": "object",
|
|
223
|
+
"additionalProperties": false,
|
|
224
|
+
"required": [
|
|
225
|
+
"name",
|
|
226
|
+
"version",
|
|
227
|
+
"content_hash"
|
|
228
|
+
],
|
|
229
|
+
"properties": {
|
|
230
|
+
"name": {
|
|
231
|
+
"type": "string",
|
|
232
|
+
"minLength": 1
|
|
233
|
+
},
|
|
234
|
+
"version": {
|
|
235
|
+
"type": "string",
|
|
236
|
+
"minLength": 1
|
|
237
|
+
},
|
|
238
|
+
"content_hash": {
|
|
239
|
+
"type": [
|
|
240
|
+
"string",
|
|
241
|
+
"null"
|
|
242
|
+
],
|
|
243
|
+
"description": "sha256 (hex) over SKILL.md + all bundled files in canonical path-sorted order. Null ONLY on an imported (DECLARED) receipt whose skill bytes were never seen; a TESTED receipt must carry the hash (see the TESTED tightening).",
|
|
244
|
+
"pattern": "^[a-f0-9]{64}$"
|
|
245
|
+
},
|
|
246
|
+
"tokens": {
|
|
247
|
+
"type": "integer",
|
|
248
|
+
"minimum": 0,
|
|
249
|
+
"description": "v0.3.1 (optional). Estimated token size of the skill's SKILL.md (a coarse chars/4 proxy, not a model tokenizer), used for the value-per-token axis (delta per 1k skill tokens). See docs/methodology.html."
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
},
|
|
253
|
+
"suite": {
|
|
254
|
+
"type": "object",
|
|
255
|
+
"additionalProperties": false,
|
|
256
|
+
"required": [
|
|
257
|
+
"format",
|
|
258
|
+
"suite_hash",
|
|
259
|
+
"case_count"
|
|
260
|
+
],
|
|
261
|
+
"properties": {
|
|
262
|
+
"format": {
|
|
263
|
+
"type": "string",
|
|
264
|
+
"minLength": 1,
|
|
265
|
+
"description": "Suite format. Driftproof-run (TESTED) receipts are always 'agentskills.io/evals' (see the TESTED tightening); an imported receipt names the source tool's format (e.g. 'skillgrade/eval.yaml')."
|
|
266
|
+
},
|
|
267
|
+
"suite_hash": {
|
|
268
|
+
"type": [
|
|
269
|
+
"string",
|
|
270
|
+
"null"
|
|
271
|
+
],
|
|
272
|
+
"description": "sha256 (hex) over the canonicalized normalized case list. Null ONLY on an imported (DECLARED) receipt whose suite bytes were never seen.",
|
|
273
|
+
"pattern": "^[a-f0-9]{64}$"
|
|
274
|
+
},
|
|
275
|
+
"case_count": {
|
|
276
|
+
"type": "integer",
|
|
277
|
+
"minimum": 0
|
|
278
|
+
},
|
|
279
|
+
"canary": {
|
|
280
|
+
"type": "string",
|
|
281
|
+
"pattern": "^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$",
|
|
282
|
+
"description": "Per-suite canary GUID: a leaked suite is detectable in a corpus. Detection aid, not a control."
|
|
283
|
+
}
|
|
284
|
+
}
|
|
285
|
+
},
|
|
286
|
+
"run": {
|
|
287
|
+
"type": "object",
|
|
288
|
+
"additionalProperties": false,
|
|
289
|
+
"required": [
|
|
290
|
+
"model_id",
|
|
291
|
+
"provider",
|
|
292
|
+
"surface",
|
|
293
|
+
"runner_version",
|
|
294
|
+
"date_utc",
|
|
295
|
+
"judge",
|
|
296
|
+
"registry",
|
|
297
|
+
"transcripts"
|
|
298
|
+
],
|
|
299
|
+
"properties": {
|
|
300
|
+
"model_id": {
|
|
301
|
+
"type": "string",
|
|
302
|
+
"minLength": 1
|
|
303
|
+
},
|
|
304
|
+
"model_release_date": {
|
|
305
|
+
"description": "ISO date (YYYY-MM-DD) of the model release if known, else null.",
|
|
306
|
+
"type": [
|
|
307
|
+
"string",
|
|
308
|
+
"null"
|
|
309
|
+
],
|
|
310
|
+
"pattern": "^\\d{4}-\\d{2}-\\d{2}$"
|
|
311
|
+
},
|
|
312
|
+
"provider": {
|
|
313
|
+
"type": "string",
|
|
314
|
+
"description": "v0.3.1. The two-axis provider the target model ran on (registry `provider`, else inferred from the id).",
|
|
315
|
+
"enum": [
|
|
316
|
+
"anthropic",
|
|
317
|
+
"openai"
|
|
318
|
+
]
|
|
319
|
+
},
|
|
320
|
+
"surface": {
|
|
321
|
+
"type": "string",
|
|
322
|
+
"enum": [
|
|
323
|
+
"api",
|
|
324
|
+
"claude-cli",
|
|
325
|
+
"openai-api",
|
|
326
|
+
"openai-cli",
|
|
327
|
+
"external"
|
|
328
|
+
],
|
|
329
|
+
"description": "'external' = the run happened on another tool's harness and was imported (never valid on a TESTED receipt)."
|
|
330
|
+
},
|
|
331
|
+
"source": {
|
|
332
|
+
"type": "string",
|
|
333
|
+
"minLength": 1,
|
|
334
|
+
"description": "Interop-additive (optional). Provenance of a converted receipt, e.g. 'imported/agent-skills-eval' or 'imported/skillgrade'. Absent on receipts Driftproof ran itself."
|
|
335
|
+
},
|
|
336
|
+
"surface_overhead_note": {
|
|
337
|
+
"type": "string",
|
|
338
|
+
"description": "v0.3.1 (optional). Present on the openai/cli surface: states the fixed Codex base-instruction preamble (~12–15k input tokens per call) that the harness prepends and does not control."
|
|
339
|
+
},
|
|
340
|
+
"status": {
|
|
341
|
+
"type": "string",
|
|
342
|
+
"description": "v0.3.1 (optional; default 'complete'). 'incomplete' when >=1 case persistently failed (e.g. failed_timeout) and was EXCLUDED from aggregates. A drift/durability report must not compute a verdict from an incomplete receipt.",
|
|
343
|
+
"enum": [
|
|
344
|
+
"complete",
|
|
345
|
+
"incomplete"
|
|
346
|
+
]
|
|
347
|
+
},
|
|
348
|
+
"failed_case_count": {
|
|
349
|
+
"type": "integer",
|
|
350
|
+
"minimum": 0,
|
|
351
|
+
"description": "v0.3.1 (optional; redefined in v0.5). The number of CASES excluded from the aggregates: exactly the length of results.aggregates.excluded_cases, computed from that list so the two cannot disagree. Exclusion is PAIRWISE since v0.5 — a case with any unmeasured arm leaves BOTH arms together and is counted ONCE here, however many of its arms failed."
|
|
352
|
+
},
|
|
353
|
+
"runner_version": {
|
|
354
|
+
"type": "string",
|
|
355
|
+
"minLength": 1
|
|
356
|
+
},
|
|
357
|
+
"date_utc": {
|
|
358
|
+
"type": "string",
|
|
359
|
+
"description": "ISO 8601 UTC timestamp of when the run finished.",
|
|
360
|
+
"pattern": "^\\d{4}-\\d{2}-\\d{2}T\\d{2}:\\d{2}:\\d{2}"
|
|
361
|
+
},
|
|
362
|
+
"registry": {
|
|
363
|
+
"type": "string",
|
|
364
|
+
"description": "Whether model_id resolved in the model registry (config/models.json). 'unregistered' means the run still executed but the model was unknown, so cost estimates used the conservative default price.",
|
|
365
|
+
"enum": [
|
|
366
|
+
"registered",
|
|
367
|
+
"unregistered"
|
|
368
|
+
]
|
|
369
|
+
},
|
|
370
|
+
"transcripts": {
|
|
371
|
+
"type": "string",
|
|
372
|
+
"description": "Transcript retention for this run. 'hashes-only' = only the sha256 hashes in results.cases are kept (the default). 'retained-local' = the raw generations + judge outputs were also written to transcripts/<receipt-id>/ (gitignored by default). 'none' = nothing retained, not even hashes — ONLY honest on an imported (DECLARED) receipt.",
|
|
373
|
+
"enum": [
|
|
374
|
+
"retained-local",
|
|
375
|
+
"hashes-only",
|
|
376
|
+
"none"
|
|
377
|
+
]
|
|
378
|
+
},
|
|
379
|
+
"judge": {
|
|
380
|
+
"type": "object",
|
|
381
|
+
"additionalProperties": false,
|
|
382
|
+
"description": "Judge sampling settings for this run.",
|
|
383
|
+
"required": [
|
|
384
|
+
"samples",
|
|
385
|
+
"temperature",
|
|
386
|
+
"sampling"
|
|
387
|
+
],
|
|
388
|
+
"properties": {
|
|
389
|
+
"samples": {
|
|
390
|
+
"type": "integer",
|
|
391
|
+
"minimum": 1,
|
|
392
|
+
"description": "Judge samples taken per case."
|
|
393
|
+
},
|
|
394
|
+
"temperature": {
|
|
395
|
+
"type": [
|
|
396
|
+
"number",
|
|
397
|
+
"null"
|
|
398
|
+
],
|
|
399
|
+
"description": "Judge temperature when the surface allows setting it (api -> 0), else null (cli -> surface-controlled)."
|
|
400
|
+
},
|
|
401
|
+
"sampling": {
|
|
402
|
+
"type": "string",
|
|
403
|
+
"description": "How sampling params were controlled, e.g. 'api-temperature-0' or 'surface-controlled'."
|
|
404
|
+
},
|
|
405
|
+
"surface": {
|
|
406
|
+
"type": "string",
|
|
407
|
+
"enum": [
|
|
408
|
+
"api",
|
|
409
|
+
"claude-cli",
|
|
410
|
+
"openai-api",
|
|
411
|
+
"openai-cli",
|
|
412
|
+
"external"
|
|
413
|
+
]
|
|
414
|
+
}
|
|
415
|
+
}
|
|
416
|
+
},
|
|
417
|
+
"pricing_snapshot": {
|
|
418
|
+
"type": "object",
|
|
419
|
+
"additionalProperties": false,
|
|
420
|
+
"description": "v0.4: registry prices FROZEN at run time. Every derived dollar figure in `economics` is computed from this snapshot and never from the live registry, so the receipt stays reproducible when registry prices later change.",
|
|
421
|
+
"required": [
|
|
422
|
+
"frozen_at",
|
|
423
|
+
"source",
|
|
424
|
+
"currency",
|
|
425
|
+
"models"
|
|
426
|
+
],
|
|
427
|
+
"properties": {
|
|
428
|
+
"frozen_at": {
|
|
429
|
+
"type": "string"
|
|
430
|
+
},
|
|
431
|
+
"source": {
|
|
432
|
+
"type": "string"
|
|
433
|
+
},
|
|
434
|
+
"currency": {
|
|
435
|
+
"type": "string"
|
|
436
|
+
},
|
|
437
|
+
"note": {
|
|
438
|
+
"type": "string"
|
|
439
|
+
},
|
|
440
|
+
"models": {
|
|
441
|
+
"type": "object",
|
|
442
|
+
"additionalProperties": {
|
|
443
|
+
"type": "object",
|
|
444
|
+
"additionalProperties": false,
|
|
445
|
+
"required": [
|
|
446
|
+
"input_per_mtok",
|
|
447
|
+
"output_per_mtok",
|
|
448
|
+
"registered"
|
|
449
|
+
],
|
|
450
|
+
"properties": {
|
|
451
|
+
"input_per_mtok": {
|
|
452
|
+
"type": "number"
|
|
453
|
+
},
|
|
454
|
+
"output_per_mtok": {
|
|
455
|
+
"type": "number"
|
|
456
|
+
},
|
|
457
|
+
"registered": {
|
|
458
|
+
"type": "boolean"
|
|
459
|
+
}
|
|
460
|
+
}
|
|
461
|
+
}
|
|
462
|
+
}
|
|
463
|
+
}
|
|
464
|
+
}
|
|
465
|
+
}
|
|
466
|
+
},
|
|
467
|
+
"results": {
|
|
468
|
+
"type": "object",
|
|
469
|
+
"additionalProperties": false,
|
|
470
|
+
"required": [
|
|
471
|
+
"cases",
|
|
472
|
+
"aggregates"
|
|
473
|
+
],
|
|
474
|
+
"properties": {
|
|
475
|
+
"cases": {
|
|
476
|
+
"type": "array",
|
|
477
|
+
"items": {
|
|
478
|
+
"type": "object",
|
|
479
|
+
"additionalProperties": false,
|
|
480
|
+
"required": [
|
|
481
|
+
"id",
|
|
482
|
+
"mode"
|
|
483
|
+
],
|
|
484
|
+
"allOf": [
|
|
485
|
+
{
|
|
486
|
+
"description": "A completed case carries the full sampled band + hashes; a failed_timeout case is recorded WITHOUT fabricated samples (it is excluded from aggregates).",
|
|
487
|
+
"if": {
|
|
488
|
+
"required": [
|
|
489
|
+
"case_status"
|
|
490
|
+
],
|
|
491
|
+
"properties": {
|
|
492
|
+
"case_status": {
|
|
493
|
+
"const": "failed_timeout"
|
|
494
|
+
}
|
|
495
|
+
}
|
|
496
|
+
},
|
|
497
|
+
"then": {
|
|
498
|
+
"required": [
|
|
499
|
+
"id",
|
|
500
|
+
"mode",
|
|
501
|
+
"case_status"
|
|
502
|
+
]
|
|
503
|
+
},
|
|
504
|
+
"else": {
|
|
505
|
+
"required": [
|
|
506
|
+
"outcome",
|
|
507
|
+
"score",
|
|
508
|
+
"mean",
|
|
509
|
+
"stddev",
|
|
510
|
+
"samples",
|
|
511
|
+
"judge"
|
|
512
|
+
]
|
|
513
|
+
}
|
|
514
|
+
}
|
|
515
|
+
],
|
|
516
|
+
"properties": {
|
|
517
|
+
"id": {
|
|
518
|
+
"type": "string",
|
|
519
|
+
"minLength": 1
|
|
520
|
+
},
|
|
521
|
+
"mode": {
|
|
522
|
+
"type": "string",
|
|
523
|
+
"enum": [
|
|
524
|
+
"with_skill",
|
|
525
|
+
"baseline"
|
|
526
|
+
]
|
|
527
|
+
},
|
|
528
|
+
"case_status": {
|
|
529
|
+
"type": "string",
|
|
530
|
+
"description": "v0.3.1 (optional; default 'ok'). 'failed_timeout' = the case's model/judge call persistently timed out after retries; the case is recorded but EXCLUDED from aggregates/verdicts — no samples/hashes are fabricated for it.",
|
|
531
|
+
"enum": [
|
|
532
|
+
"ok",
|
|
533
|
+
"failed_timeout"
|
|
534
|
+
]
|
|
535
|
+
},
|
|
536
|
+
"outcome": {
|
|
537
|
+
"type": "string",
|
|
538
|
+
"description": "borderline = the threshold lies within mean +/- stddev.",
|
|
539
|
+
"enum": [
|
|
540
|
+
"pass",
|
|
541
|
+
"fail",
|
|
542
|
+
"borderline",
|
|
543
|
+
"score"
|
|
544
|
+
]
|
|
545
|
+
},
|
|
546
|
+
"score": {
|
|
547
|
+
"type": "number",
|
|
548
|
+
"minimum": 0,
|
|
549
|
+
"maximum": 1,
|
|
550
|
+
"description": "Alias of mean, kept for v0.1 readers."
|
|
551
|
+
},
|
|
552
|
+
"mean": {
|
|
553
|
+
"type": "number",
|
|
554
|
+
"minimum": 0,
|
|
555
|
+
"maximum": 1
|
|
556
|
+
},
|
|
557
|
+
"stddev": {
|
|
558
|
+
"type": "number",
|
|
559
|
+
"minimum": 0,
|
|
560
|
+
"description": "Sample stddev of the judge samples (raw band half-width)."
|
|
561
|
+
},
|
|
562
|
+
"samples": {
|
|
563
|
+
"type": "array",
|
|
564
|
+
"items": {
|
|
565
|
+
"type": "number",
|
|
566
|
+
"minimum": 0,
|
|
567
|
+
"maximum": 1
|
|
568
|
+
},
|
|
569
|
+
"minItems": 1
|
|
570
|
+
},
|
|
571
|
+
"generation_hash": {
|
|
572
|
+
"type": "string",
|
|
573
|
+
"description": "sha256 (hex) of the raw model generation that was judged for this (case, mode).",
|
|
574
|
+
"pattern": "^[a-f0-9]{64}$"
|
|
575
|
+
},
|
|
576
|
+
"judge_sample_hashes": {
|
|
577
|
+
"type": "array",
|
|
578
|
+
"description": "sha256 (hex) of each raw judge output, one per judge sample. Same length as `samples`.",
|
|
579
|
+
"items": {
|
|
580
|
+
"type": "string",
|
|
581
|
+
"pattern": "^[a-f0-9]{64}$"
|
|
582
|
+
},
|
|
583
|
+
"minItems": 1
|
|
584
|
+
},
|
|
585
|
+
"threshold": {
|
|
586
|
+
"type": [
|
|
587
|
+
"number",
|
|
588
|
+
"null"
|
|
589
|
+
],
|
|
590
|
+
"minimum": 0,
|
|
591
|
+
"maximum": 1
|
|
592
|
+
},
|
|
593
|
+
"reason": {
|
|
594
|
+
"type": "string"
|
|
595
|
+
},
|
|
596
|
+
"checks": {
|
|
597
|
+
"type": "array",
|
|
598
|
+
"description": "v0.3.1 (optional). Deterministic post-check results for this (case, mode): structural/regex assertions run on the model output ALONGSIDE the judge. Supplementary evidence reported as a separate column — NOT folded into the outcome/band verdict.",
|
|
599
|
+
"items": {
|
|
600
|
+
"type": "object",
|
|
601
|
+
"additionalProperties": false,
|
|
602
|
+
"required": [
|
|
603
|
+
"name",
|
|
604
|
+
"kind",
|
|
605
|
+
"pass"
|
|
606
|
+
],
|
|
607
|
+
"properties": {
|
|
608
|
+
"name": {
|
|
609
|
+
"type": "string",
|
|
610
|
+
"minLength": 1
|
|
611
|
+
},
|
|
612
|
+
"kind": {
|
|
613
|
+
"type": "string",
|
|
614
|
+
"enum": [
|
|
615
|
+
"regex",
|
|
616
|
+
"contains",
|
|
617
|
+
"not_contains",
|
|
618
|
+
"min_length"
|
|
619
|
+
]
|
|
620
|
+
},
|
|
621
|
+
"pass": {
|
|
622
|
+
"type": "boolean"
|
|
623
|
+
}
|
|
624
|
+
}
|
|
625
|
+
}
|
|
626
|
+
},
|
|
627
|
+
"judge": {
|
|
628
|
+
"type": "object",
|
|
629
|
+
"additionalProperties": false,
|
|
630
|
+
"required": [
|
|
631
|
+
"model_id",
|
|
632
|
+
"rubric_hash"
|
|
633
|
+
],
|
|
634
|
+
"properties": {
|
|
635
|
+
"model_id": {
|
|
636
|
+
"type": "string",
|
|
637
|
+
"minLength": 1
|
|
638
|
+
},
|
|
639
|
+
"rubric_hash": {
|
|
640
|
+
"type": [
|
|
641
|
+
"string",
|
|
642
|
+
"null"
|
|
643
|
+
],
|
|
644
|
+
"description": "Null ONLY on an imported (DECLARED) receipt whose rubric bytes were never seen.",
|
|
645
|
+
"pattern": "^[a-f0-9]{64}$"
|
|
646
|
+
}
|
|
647
|
+
}
|
|
648
|
+
},
|
|
649
|
+
"usage": {
|
|
650
|
+
"type": "object",
|
|
651
|
+
"additionalProperties": false,
|
|
652
|
+
"description": "v0.4: usage of the GENERATION call for this (case, mode) row. Normalized usage for ONE call. input_tokens is the TOTAL input presented to the model INCLUDING any cached portion (the surfaces disagree about this natively; see lib/usage.js). cached_tokens is the portion served from cache, null when the surface does not report it. output_tokens includes reasoning/thinking tokens where the surface bundles them. wall_ms is measured by the runner around the successful attempt, so it means the same thing on every surface. A field the surface did not report is null — never 0.",
|
|
653
|
+
"required": [
|
|
654
|
+
"input_tokens",
|
|
655
|
+
"output_tokens",
|
|
656
|
+
"cached_tokens",
|
|
657
|
+
"wall_ms"
|
|
658
|
+
],
|
|
659
|
+
"properties": {
|
|
660
|
+
"input_tokens": {
|
|
661
|
+
"type": [
|
|
662
|
+
"integer",
|
|
663
|
+
"null"
|
|
664
|
+
],
|
|
665
|
+
"minimum": 0
|
|
666
|
+
},
|
|
667
|
+
"output_tokens": {
|
|
668
|
+
"type": [
|
|
669
|
+
"integer",
|
|
670
|
+
"null"
|
|
671
|
+
],
|
|
672
|
+
"minimum": 0
|
|
673
|
+
},
|
|
674
|
+
"cached_tokens": {
|
|
675
|
+
"type": [
|
|
676
|
+
"integer",
|
|
677
|
+
"null"
|
|
678
|
+
],
|
|
679
|
+
"minimum": 0
|
|
680
|
+
},
|
|
681
|
+
"wall_ms": {
|
|
682
|
+
"type": [
|
|
683
|
+
"integer",
|
|
684
|
+
"null"
|
|
685
|
+
],
|
|
686
|
+
"minimum": 0
|
|
687
|
+
}
|
|
688
|
+
}
|
|
689
|
+
},
|
|
690
|
+
"judge_usage": {
|
|
691
|
+
"type": "object",
|
|
692
|
+
"additionalProperties": false,
|
|
693
|
+
"description": "v0.4: SUM of usage over the N judge calls that graded this row. Measurement overhead imposed by the harness, NOT a cost of running the skill — excluded from every field in `economics` by construction.",
|
|
694
|
+
"required": [
|
|
695
|
+
"input_tokens",
|
|
696
|
+
"output_tokens",
|
|
697
|
+
"cached_tokens",
|
|
698
|
+
"wall_ms"
|
|
699
|
+
],
|
|
700
|
+
"properties": {
|
|
701
|
+
"input_tokens": {
|
|
702
|
+
"type": [
|
|
703
|
+
"integer",
|
|
704
|
+
"null"
|
|
705
|
+
],
|
|
706
|
+
"minimum": 0
|
|
707
|
+
},
|
|
708
|
+
"output_tokens": {
|
|
709
|
+
"type": [
|
|
710
|
+
"integer",
|
|
711
|
+
"null"
|
|
712
|
+
],
|
|
713
|
+
"minimum": 0
|
|
714
|
+
},
|
|
715
|
+
"cached_tokens": {
|
|
716
|
+
"type": [
|
|
717
|
+
"integer",
|
|
718
|
+
"null"
|
|
719
|
+
],
|
|
720
|
+
"minimum": 0
|
|
721
|
+
},
|
|
722
|
+
"wall_ms": {
|
|
723
|
+
"type": [
|
|
724
|
+
"integer",
|
|
725
|
+
"null"
|
|
726
|
+
],
|
|
727
|
+
"minimum": 0
|
|
728
|
+
}
|
|
729
|
+
}
|
|
730
|
+
},
|
|
731
|
+
"generation": {
|
|
732
|
+
"type": "object",
|
|
733
|
+
"additionalProperties": true,
|
|
734
|
+
"required": [
|
|
735
|
+
"n_drawn",
|
|
736
|
+
"n_measured",
|
|
737
|
+
"n_unmeasured",
|
|
738
|
+
"stopping_reason",
|
|
739
|
+
"draws"
|
|
740
|
+
],
|
|
741
|
+
"properties": {
|
|
742
|
+
"n_planned": {
|
|
743
|
+
"type": "integer",
|
|
744
|
+
"minimum": 1
|
|
745
|
+
},
|
|
746
|
+
"n_drawn": {
|
|
747
|
+
"type": "integer",
|
|
748
|
+
"minimum": 0
|
|
749
|
+
},
|
|
750
|
+
"n_measured": {
|
|
751
|
+
"type": "integer",
|
|
752
|
+
"minimum": 0
|
|
753
|
+
},
|
|
754
|
+
"n_unmeasured": {
|
|
755
|
+
"type": "integer",
|
|
756
|
+
"minimum": 0
|
|
757
|
+
},
|
|
758
|
+
"stopping_reason": {
|
|
759
|
+
"enum": [
|
|
760
|
+
"min_reached",
|
|
761
|
+
"stabilised",
|
|
762
|
+
"max_reached",
|
|
763
|
+
"unmeasured_exhausted",
|
|
764
|
+
"below_min",
|
|
765
|
+
"escalating"
|
|
766
|
+
]
|
|
767
|
+
},
|
|
768
|
+
"mean": {
|
|
769
|
+
"type": [
|
|
770
|
+
"number",
|
|
771
|
+
"null"
|
|
772
|
+
]
|
|
773
|
+
},
|
|
774
|
+
"sd": {
|
|
775
|
+
"type": [
|
|
776
|
+
"number",
|
|
777
|
+
"null"
|
|
778
|
+
]
|
|
779
|
+
},
|
|
780
|
+
"judge_sd_mean": {
|
|
781
|
+
"type": [
|
|
782
|
+
"number",
|
|
783
|
+
"null"
|
|
784
|
+
]
|
|
785
|
+
},
|
|
786
|
+
"variance_ratio": {
|
|
787
|
+
"type": [
|
|
788
|
+
"number",
|
|
789
|
+
"null"
|
|
790
|
+
]
|
|
791
|
+
},
|
|
792
|
+
"draws": {
|
|
793
|
+
"type": "array",
|
|
794
|
+
"items": {
|
|
795
|
+
"type": "object",
|
|
796
|
+
"additionalProperties": true,
|
|
797
|
+
"required": [
|
|
798
|
+
"draw_index",
|
|
799
|
+
"status"
|
|
800
|
+
],
|
|
801
|
+
"properties": {
|
|
802
|
+
"draw_index": {
|
|
803
|
+
"type": "integer",
|
|
804
|
+
"minimum": 0
|
|
805
|
+
},
|
|
806
|
+
"status": {
|
|
807
|
+
"enum": [
|
|
808
|
+
"measured",
|
|
809
|
+
"unmeasured"
|
|
810
|
+
]
|
|
811
|
+
},
|
|
812
|
+
"generation_hash": {
|
|
813
|
+
"type": [
|
|
814
|
+
"string",
|
|
815
|
+
"null"
|
|
816
|
+
]
|
|
817
|
+
},
|
|
818
|
+
"samples": {
|
|
819
|
+
"type": "array",
|
|
820
|
+
"items": {
|
|
821
|
+
"type": "number"
|
|
822
|
+
}
|
|
823
|
+
},
|
|
824
|
+
"judge_sample_hashes": {
|
|
825
|
+
"type": "array",
|
|
826
|
+
"items": {
|
|
827
|
+
"type": "string"
|
|
828
|
+
}
|
|
829
|
+
},
|
|
830
|
+
"mean": {
|
|
831
|
+
"type": [
|
|
832
|
+
"number",
|
|
833
|
+
"null"
|
|
834
|
+
]
|
|
835
|
+
},
|
|
836
|
+
"stddev": {
|
|
837
|
+
"type": [
|
|
838
|
+
"number",
|
|
839
|
+
"null"
|
|
840
|
+
]
|
|
841
|
+
},
|
|
842
|
+
"reason": {
|
|
843
|
+
"type": "string"
|
|
844
|
+
},
|
|
845
|
+
"usage": {
|
|
846
|
+
"type": "object"
|
|
847
|
+
},
|
|
848
|
+
"judge_usage": {
|
|
849
|
+
"type": "object"
|
|
850
|
+
}
|
|
851
|
+
},
|
|
852
|
+
"allOf": [
|
|
853
|
+
{
|
|
854
|
+
"if": {
|
|
855
|
+
"properties": {
|
|
856
|
+
"status": {
|
|
857
|
+
"const": "unmeasured"
|
|
858
|
+
}
|
|
859
|
+
},
|
|
860
|
+
"required": [
|
|
861
|
+
"status"
|
|
862
|
+
]
|
|
863
|
+
},
|
|
864
|
+
"then": {
|
|
865
|
+
"properties": {
|
|
866
|
+
"mean": {
|
|
867
|
+
"type": "null"
|
|
868
|
+
},
|
|
869
|
+
"stddev": {
|
|
870
|
+
"type": "null"
|
|
871
|
+
},
|
|
872
|
+
"samples": {
|
|
873
|
+
"maxItems": 0
|
|
874
|
+
}
|
|
875
|
+
}
|
|
876
|
+
}
|
|
877
|
+
},
|
|
878
|
+
{
|
|
879
|
+
"if": {
|
|
880
|
+
"properties": {
|
|
881
|
+
"status": {
|
|
882
|
+
"const": "measured"
|
|
883
|
+
}
|
|
884
|
+
},
|
|
885
|
+
"required": [
|
|
886
|
+
"status"
|
|
887
|
+
]
|
|
888
|
+
},
|
|
889
|
+
"then": {
|
|
890
|
+
"required": [
|
|
891
|
+
"generation_hash",
|
|
892
|
+
"samples",
|
|
893
|
+
"mean",
|
|
894
|
+
"stddev"
|
|
895
|
+
],
|
|
896
|
+
"properties": {
|
|
897
|
+
"mean": {
|
|
898
|
+
"type": "number"
|
|
899
|
+
},
|
|
900
|
+
"generation_hash": {
|
|
901
|
+
"type": "string"
|
|
902
|
+
}
|
|
903
|
+
}
|
|
904
|
+
}
|
|
905
|
+
}
|
|
906
|
+
]
|
|
907
|
+
}
|
|
908
|
+
},
|
|
909
|
+
"variance_ratio_unavailable": {
|
|
910
|
+
"type": [
|
|
911
|
+
"string",
|
|
912
|
+
"null"
|
|
913
|
+
],
|
|
914
|
+
"enum": [
|
|
915
|
+
"single_judge_sample",
|
|
916
|
+
"judge_sd_zero",
|
|
917
|
+
"no_measured_draws",
|
|
918
|
+
"judge_samples_unknown",
|
|
919
|
+
null
|
|
920
|
+
],
|
|
921
|
+
"description": "WHICH null the variance_ratio is (F-014-C). `null` when a ratio was formed. `single_judge_sample`: k=1, so the per-draw judge spread is 0 by construction and the ratio is undefined — this is the shape every pre-v0.5 example produced. `judge_sd_zero`: k>=2 and the judge agreed with itself perfectly inside every measured draw. `no_measured_draws`: nothing was measured. `judge_samples_unknown`: the draws carry no sample list, so the count cannot be established and saying which null it is would assert a cause this control cannot reach. Each names what was OBSERVED and none names a cause."
|
|
922
|
+
}
|
|
923
|
+
},
|
|
924
|
+
"allOf": [
|
|
925
|
+
{
|
|
926
|
+
"$comment": "driftproof/variance-ratio-cause",
|
|
927
|
+
"description": "F-014-C: a null ratio must say WHICH null it is. Bound to the null rather than required outright, so a receipt that formed a ratio carries no reason and nothing in the archive is retroactively incomplete.",
|
|
928
|
+
"if": {
|
|
929
|
+
"required": [
|
|
930
|
+
"variance_ratio"
|
|
931
|
+
],
|
|
932
|
+
"properties": {
|
|
933
|
+
"variance_ratio": {
|
|
934
|
+
"type": "null"
|
|
935
|
+
}
|
|
936
|
+
}
|
|
937
|
+
},
|
|
938
|
+
"then": {
|
|
939
|
+
"required": [
|
|
940
|
+
"variance_ratio_unavailable"
|
|
941
|
+
],
|
|
942
|
+
"properties": {
|
|
943
|
+
"variance_ratio_unavailable": {
|
|
944
|
+
"type": "string"
|
|
945
|
+
}
|
|
946
|
+
}
|
|
947
|
+
}
|
|
948
|
+
}
|
|
949
|
+
]
|
|
950
|
+
}
|
|
951
|
+
}
|
|
952
|
+
}
|
|
953
|
+
},
|
|
954
|
+
"aggregates": {
|
|
955
|
+
"type": "object",
|
|
956
|
+
"additionalProperties": false,
|
|
957
|
+
"required": [
|
|
958
|
+
"with_skill",
|
|
959
|
+
"baseline"
|
|
960
|
+
],
|
|
961
|
+
"properties": {
|
|
962
|
+
"with_skill": {
|
|
963
|
+
"$ref": "#/$defs/modeAggregate"
|
|
964
|
+
},
|
|
965
|
+
"baseline": {
|
|
966
|
+
"$ref": "#/$defs/modeAggregate"
|
|
967
|
+
},
|
|
968
|
+
"excluded_cases": {
|
|
969
|
+
"type": "array",
|
|
970
|
+
"description": "Cases excluded from BOTH arms of the aggregate because at least one of their arms had no measured result (spec 017 AC-6). `comparison.delta` is paired by construction, so a case that cannot be measured on one side is removed from both rather than from one — otherwise the delta is a mean over one case set minus a mean over another. The case itself remains in results.cases: it is removed from the mean, not from the record. Absent when nothing was excluded.",
|
|
971
|
+
"items": {
|
|
972
|
+
"type": "object",
|
|
973
|
+
"additionalProperties": true,
|
|
974
|
+
"required": [
|
|
975
|
+
"id",
|
|
976
|
+
"reason"
|
|
977
|
+
],
|
|
978
|
+
"properties": {
|
|
979
|
+
"id": {
|
|
980
|
+
"type": "string",
|
|
981
|
+
"description": "The case id excluded from both arms."
|
|
982
|
+
},
|
|
983
|
+
"modes": {
|
|
984
|
+
"type": "array",
|
|
985
|
+
"items": {
|
|
986
|
+
"enum": [
|
|
987
|
+
"with_skill",
|
|
988
|
+
"baseline"
|
|
989
|
+
]
|
|
990
|
+
},
|
|
991
|
+
"description": "Which arm or arms were unusable."
|
|
992
|
+
},
|
|
993
|
+
"reason": {
|
|
994
|
+
"type": "string",
|
|
995
|
+
"description": "What was observed. Names no cause the receipt cannot establish."
|
|
996
|
+
}
|
|
997
|
+
}
|
|
998
|
+
}
|
|
999
|
+
}
|
|
1000
|
+
}
|
|
1001
|
+
}
|
|
1002
|
+
}
|
|
1003
|
+
},
|
|
1004
|
+
"comparison": {
|
|
1005
|
+
"type": "object",
|
|
1006
|
+
"additionalProperties": false,
|
|
1007
|
+
"required": [
|
|
1008
|
+
"with_skill_score",
|
|
1009
|
+
"baseline_score",
|
|
1010
|
+
"delta",
|
|
1011
|
+
"delta_uncertainty"
|
|
1012
|
+
],
|
|
1013
|
+
"properties": {
|
|
1014
|
+
"with_skill_score": {
|
|
1015
|
+
"type": "number",
|
|
1016
|
+
"minimum": 0,
|
|
1017
|
+
"maximum": 1
|
|
1018
|
+
},
|
|
1019
|
+
"baseline_score": {
|
|
1020
|
+
"type": [
|
|
1021
|
+
"number",
|
|
1022
|
+
"null"
|
|
1023
|
+
],
|
|
1024
|
+
"minimum": 0,
|
|
1025
|
+
"maximum": 1,
|
|
1026
|
+
"description": "Null ONLY on an imported (DECLARED) receipt from a tool with no baseline mode — never a fabricated 0."
|
|
1027
|
+
},
|
|
1028
|
+
"delta": {
|
|
1029
|
+
"type": [
|
|
1030
|
+
"number",
|
|
1031
|
+
"null"
|
|
1032
|
+
],
|
|
1033
|
+
"minimum": -1,
|
|
1034
|
+
"maximum": 1,
|
|
1035
|
+
"description": "Null when baseline_score is null (no baseline mode was run)."
|
|
1036
|
+
},
|
|
1037
|
+
"delta_uncertainty": {
|
|
1038
|
+
"type": [
|
|
1039
|
+
"number",
|
|
1040
|
+
"null"
|
|
1041
|
+
],
|
|
1042
|
+
"minimum": 0,
|
|
1043
|
+
"description": "Combined uncertainty of the delta (quadrature sum of the two aggregate bands). Null when delta is null."
|
|
1044
|
+
}
|
|
1045
|
+
}
|
|
1046
|
+
},
|
|
1047
|
+
"verification_level": {
|
|
1048
|
+
"type": "string",
|
|
1049
|
+
"description": "Community verification lattice. FORMAL is reserved/unimplemented in v0.3.",
|
|
1050
|
+
"enum": [
|
|
1051
|
+
"UNVERIFIED",
|
|
1052
|
+
"DECLARED",
|
|
1053
|
+
"TESTED"
|
|
1054
|
+
]
|
|
1055
|
+
},
|
|
1056
|
+
"editorial_reviews": {
|
|
1057
|
+
"type": "array",
|
|
1058
|
+
"description": "Optional pointers to external one-shot editorial reviews of this skill (context only; not verification evidence).",
|
|
1059
|
+
"items": {
|
|
1060
|
+
"type": "object",
|
|
1061
|
+
"additionalProperties": false,
|
|
1062
|
+
"required": [
|
|
1063
|
+
"url",
|
|
1064
|
+
"source",
|
|
1065
|
+
"date"
|
|
1066
|
+
],
|
|
1067
|
+
"properties": {
|
|
1068
|
+
"url": {
|
|
1069
|
+
"type": "string",
|
|
1070
|
+
"minLength": 1
|
|
1071
|
+
},
|
|
1072
|
+
"source": {
|
|
1073
|
+
"type": "string",
|
|
1074
|
+
"minLength": 1
|
|
1075
|
+
},
|
|
1076
|
+
"date": {
|
|
1077
|
+
"type": "string",
|
|
1078
|
+
"pattern": "^\\d{4}-\\d{2}-\\d{2}$"
|
|
1079
|
+
}
|
|
1080
|
+
}
|
|
1081
|
+
}
|
|
1082
|
+
},
|
|
1083
|
+
"receipt_hash": {
|
|
1084
|
+
"type": "string",
|
|
1085
|
+
"description": "sha256 (hex) of the canonical receipt JSON with this field omitted.",
|
|
1086
|
+
"pattern": "^[a-f0-9]{64}$"
|
|
1087
|
+
},
|
|
1088
|
+
"economics": {
|
|
1089
|
+
"type": "object",
|
|
1090
|
+
"additionalProperties": false,
|
|
1091
|
+
"description": "v0.4 derived economics. Computed from the per-case `usage` at `run.pricing_snapshot` prices. The three value axes (accuracy lift, cost, latency) live separately here and in the reports: there is deliberately NO composite value score, because collapsing axes with different units and different error bars would produce a number no reader could trace to evidence.",
|
|
1092
|
+
"required": [
|
|
1093
|
+
"basis",
|
|
1094
|
+
"with_skill",
|
|
1095
|
+
"baseline",
|
|
1096
|
+
"judge_excluded"
|
|
1097
|
+
],
|
|
1098
|
+
"properties": {
|
|
1099
|
+
"basis": {
|
|
1100
|
+
"enum": [
|
|
1101
|
+
"metered",
|
|
1102
|
+
"metered-equivalent"
|
|
1103
|
+
],
|
|
1104
|
+
"description": "metered = real spend on an api surface; metered-equivalent = what the same tokens would have cost on the metered API (subscription CLI surfaces, where actual spend is $0)."
|
|
1105
|
+
},
|
|
1106
|
+
"surface": {
|
|
1107
|
+
"type": "string"
|
|
1108
|
+
},
|
|
1109
|
+
"with_skill": {
|
|
1110
|
+
"type": "object",
|
|
1111
|
+
"additionalProperties": false,
|
|
1112
|
+
"properties": {
|
|
1113
|
+
"call_count": {
|
|
1114
|
+
"type": "integer",
|
|
1115
|
+
"minimum": 0
|
|
1116
|
+
},
|
|
1117
|
+
"mean_input_tokens": {
|
|
1118
|
+
"type": [
|
|
1119
|
+
"number",
|
|
1120
|
+
"null"
|
|
1121
|
+
]
|
|
1122
|
+
},
|
|
1123
|
+
"mean_output_tokens": {
|
|
1124
|
+
"type": [
|
|
1125
|
+
"number",
|
|
1126
|
+
"null"
|
|
1127
|
+
]
|
|
1128
|
+
},
|
|
1129
|
+
"mean_cost_usd_per_call": {
|
|
1130
|
+
"type": [
|
|
1131
|
+
"number",
|
|
1132
|
+
"null"
|
|
1133
|
+
]
|
|
1134
|
+
},
|
|
1135
|
+
"median_wall_ms": {
|
|
1136
|
+
"type": [
|
|
1137
|
+
"number",
|
|
1138
|
+
"null"
|
|
1139
|
+
]
|
|
1140
|
+
},
|
|
1141
|
+
"wall_ms_p25": {
|
|
1142
|
+
"type": [
|
|
1143
|
+
"number",
|
|
1144
|
+
"null"
|
|
1145
|
+
]
|
|
1146
|
+
},
|
|
1147
|
+
"wall_ms_p75": {
|
|
1148
|
+
"type": [
|
|
1149
|
+
"number",
|
|
1150
|
+
"null"
|
|
1151
|
+
]
|
|
1152
|
+
},
|
|
1153
|
+
"wall_ms_iqr": {
|
|
1154
|
+
"type": [
|
|
1155
|
+
"number",
|
|
1156
|
+
"null"
|
|
1157
|
+
]
|
|
1158
|
+
}
|
|
1159
|
+
}
|
|
1160
|
+
},
|
|
1161
|
+
"baseline": {
|
|
1162
|
+
"type": "object",
|
|
1163
|
+
"additionalProperties": false,
|
|
1164
|
+
"properties": {
|
|
1165
|
+
"call_count": {
|
|
1166
|
+
"type": "integer",
|
|
1167
|
+
"minimum": 0
|
|
1168
|
+
},
|
|
1169
|
+
"mean_input_tokens": {
|
|
1170
|
+
"type": [
|
|
1171
|
+
"number",
|
|
1172
|
+
"null"
|
|
1173
|
+
]
|
|
1174
|
+
},
|
|
1175
|
+
"mean_output_tokens": {
|
|
1176
|
+
"type": [
|
|
1177
|
+
"number",
|
|
1178
|
+
"null"
|
|
1179
|
+
]
|
|
1180
|
+
},
|
|
1181
|
+
"mean_cost_usd_per_call": {
|
|
1182
|
+
"type": [
|
|
1183
|
+
"number",
|
|
1184
|
+
"null"
|
|
1185
|
+
]
|
|
1186
|
+
},
|
|
1187
|
+
"median_wall_ms": {
|
|
1188
|
+
"type": [
|
|
1189
|
+
"number",
|
|
1190
|
+
"null"
|
|
1191
|
+
]
|
|
1192
|
+
},
|
|
1193
|
+
"wall_ms_p25": {
|
|
1194
|
+
"type": [
|
|
1195
|
+
"number",
|
|
1196
|
+
"null"
|
|
1197
|
+
]
|
|
1198
|
+
},
|
|
1199
|
+
"wall_ms_p75": {
|
|
1200
|
+
"type": [
|
|
1201
|
+
"number",
|
|
1202
|
+
"null"
|
|
1203
|
+
]
|
|
1204
|
+
},
|
|
1205
|
+
"wall_ms_iqr": {
|
|
1206
|
+
"type": [
|
|
1207
|
+
"number",
|
|
1208
|
+
"null"
|
|
1209
|
+
]
|
|
1210
|
+
}
|
|
1211
|
+
}
|
|
1212
|
+
},
|
|
1213
|
+
"skill_incremental_cost_usd_per_call": {
|
|
1214
|
+
"type": [
|
|
1215
|
+
"number",
|
|
1216
|
+
"null"
|
|
1217
|
+
]
|
|
1218
|
+
},
|
|
1219
|
+
"skill_incremental_cost_usd_per_1k_calls": {
|
|
1220
|
+
"type": [
|
|
1221
|
+
"number",
|
|
1222
|
+
"null"
|
|
1223
|
+
]
|
|
1224
|
+
},
|
|
1225
|
+
"output_tokens_delta": {
|
|
1226
|
+
"type": [
|
|
1227
|
+
"number",
|
|
1228
|
+
"null"
|
|
1229
|
+
]
|
|
1230
|
+
},
|
|
1231
|
+
"median_wall_ms_delta": {
|
|
1232
|
+
"type": [
|
|
1233
|
+
"number",
|
|
1234
|
+
"null"
|
|
1235
|
+
]
|
|
1236
|
+
},
|
|
1237
|
+
"judge_excluded": {
|
|
1238
|
+
"const": true,
|
|
1239
|
+
"description": "Structural guarantee: judge usage never enters any figure in this block. A receipt cannot claim otherwise."
|
|
1240
|
+
},
|
|
1241
|
+
"judge_overhead": {
|
|
1242
|
+
"type": "object",
|
|
1243
|
+
"additionalProperties": false,
|
|
1244
|
+
"properties": {
|
|
1245
|
+
"note": {
|
|
1246
|
+
"type": "string"
|
|
1247
|
+
},
|
|
1248
|
+
"total_cost_usd": {
|
|
1249
|
+
"type": [
|
|
1250
|
+
"number",
|
|
1251
|
+
"null"
|
|
1252
|
+
]
|
|
1253
|
+
},
|
|
1254
|
+
"case_rows_measured": {
|
|
1255
|
+
"type": "integer",
|
|
1256
|
+
"minimum": 0
|
|
1257
|
+
}
|
|
1258
|
+
}
|
|
1259
|
+
},
|
|
1260
|
+
"notes": {
|
|
1261
|
+
"type": "object",
|
|
1262
|
+
"additionalProperties": false,
|
|
1263
|
+
"properties": {
|
|
1264
|
+
"absolute_cost": {
|
|
1265
|
+
"type": "string"
|
|
1266
|
+
},
|
|
1267
|
+
"cache_pricing": {
|
|
1268
|
+
"type": "string"
|
|
1269
|
+
},
|
|
1270
|
+
"latency": {
|
|
1271
|
+
"type": "string"
|
|
1272
|
+
}
|
|
1273
|
+
}
|
|
1274
|
+
}
|
|
1275
|
+
}
|
|
1276
|
+
},
|
|
1277
|
+
"generation_sampled": {
|
|
1278
|
+
"type": "boolean",
|
|
1279
|
+
"description": "CAPABILITY FLAG (v0.5). A receipt that carries across-draw statistics — any results.cases[] entry with a `generation` block — MUST declare `generation_sampled: true`, and a receipt that declares it MUST carry at least one AND MUST carry `suite.canary` (F-015-A: F-014-F named both blocks). ABSENT MEANS LEGACY: a v0.5 receipt that ran no generation sampling (an imported DECLARED receipt, for instance) omits this field and stays valid, and every v0.4-and-earlier receipt is unaffected. The flag exists because v0.5 otherwise let a receipt claim conformance while carrying none of what v0.5 adds (F-014-F); binding the requirement to what the receipt DECLARES rather than to its verification_level closes that without invalidating a single archived receipt."
|
|
1280
|
+
}
|
|
1281
|
+
},
|
|
1282
|
+
"$defs": {
|
|
1283
|
+
"modeAggregate": {
|
|
1284
|
+
"type": "object",
|
|
1285
|
+
"additionalProperties": false,
|
|
1286
|
+
"required": [
|
|
1287
|
+
"case_count",
|
|
1288
|
+
"pass_count",
|
|
1289
|
+
"mean_score",
|
|
1290
|
+
"stddev"
|
|
1291
|
+
],
|
|
1292
|
+
"properties": {
|
|
1293
|
+
"case_count": {
|
|
1294
|
+
"type": "integer",
|
|
1295
|
+
"minimum": 0
|
|
1296
|
+
},
|
|
1297
|
+
"pass_count": {
|
|
1298
|
+
"type": "integer",
|
|
1299
|
+
"minimum": 0
|
|
1300
|
+
},
|
|
1301
|
+
"borderline_count": {
|
|
1302
|
+
"type": "integer",
|
|
1303
|
+
"minimum": 0
|
|
1304
|
+
},
|
|
1305
|
+
"mean_score": {
|
|
1306
|
+
"type": "number",
|
|
1307
|
+
"minimum": 0,
|
|
1308
|
+
"maximum": 1
|
|
1309
|
+
},
|
|
1310
|
+
"stddev": {
|
|
1311
|
+
"type": "number",
|
|
1312
|
+
"minimum": 0,
|
|
1313
|
+
"description": "Suite dispersion: stddev of the per-case means across the suite. A reported summary stat; the drift headline is driven by per-case band-overlap verdicts, not this band."
|
|
1314
|
+
}
|
|
1315
|
+
}
|
|
1316
|
+
}
|
|
1317
|
+
}
|
|
1318
|
+
}
|