driftproof 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,45 +2,111 @@
2
2
  "$schema": "https://json-schema.org/draft/2020-12/schema",
3
3
  "$id": "https://driftproofhq.com/spec/receipt.schema.json",
4
4
  "title": "driftproof receipt",
5
- "description": "A signed, dated record of running one agent skill's eval suite with and without the skill on one model version, with sampled judge scores and confidence bands. Receipt spec v0.3.1 — additive over v0.3: adds run.provider (two-axis provider), expands the surface enum with the OpenAI lanes (openai-api, openai-cli), adds an optional run.surface_overhead_note (fixed harness preamble on the openai/cli surface), optional per-case deterministic post-check results, and an optional skill.tokens count for the value-per-token axis. Interop-additive revision: receipts IMPORTED from external tools (surface 'external', run.source 'imported/<tool>', verification_level DECLARED) may carry null content/suite/rubric hashes and omit generation hashes — hashes are never fabricated; the TESTED tightening (see allOf) requires the full evidence chain whenever verification_level is TESTED, so no previously issued receipt is invalidated and TESTED keeps its meaning.",
5
+ "description": "A hash-verified, dated record of running one agent skill's eval suite with and without the skill on one model version, with sampled judge scores and confidence bands. Receipt spec v0.4 — additive over v0.3.1: adds per-case, per-arm generation `usage` (input/output/cached tokens + measured wall_ms) captured from the surfaces that report it, a separate per-case `judge_usage` (measurement overhead, EXCLUDED from every skill-value figure by construction — `economics.judge_excluded` is const true), a run-level `run.pricing_snapshot` freezing the registry prices the derived dollar figures were computed from (so a receipt keeps its meaning when prices later change), and a derived `economics` block (per-arm mean cost/call, skill incremental cost per call and per 1k calls, output-length delta, median wall_ms with IQR). The three value axes — accuracy lift, cost, latency — are recorded separately and NEVER combined into a composite score. All v0.3.1 semantics are unchanged and every prior receipt still validates against its own frozen schema (v0.1, v0.2, v0.3, v0.3.1). The TESTED tightening (see allOf) is unchanged: the interop relaxations remain available only below TESTED.",
6
6
  "type": "object",
7
7
  "additionalProperties": false,
8
- "required": ["schema_version", "skill", "suite", "run", "results", "comparison", "verification_level", "receipt_hash"],
8
+ "required": [
9
+ "schema_version",
10
+ "skill",
11
+ "suite",
12
+ "run",
13
+ "results",
14
+ "comparison",
15
+ "verification_level",
16
+ "receipt_hash"
17
+ ],
9
18
  "allOf": [
10
19
  {
11
20
  "description": "TESTED tightening — the interop relaxations (null hashes, external surface, null comparison, transcripts 'none') are ONLY available to receipts below TESTED. A TESTED receipt must carry the full evidence chain, exactly as before the interop revision.",
12
- "if": { "required": ["verification_level"], "properties": { "verification_level": { "const": "TESTED" } } },
21
+ "if": {
22
+ "required": [
23
+ "verification_level"
24
+ ],
25
+ "properties": {
26
+ "verification_level": {
27
+ "const": "TESTED"
28
+ }
29
+ }
30
+ },
13
31
  "then": {
14
32
  "properties": {
15
- "skill": { "properties": { "content_hash": { "type": "string" } } },
33
+ "skill": {
34
+ "properties": {
35
+ "content_hash": {
36
+ "type": "string"
37
+ }
38
+ }
39
+ },
16
40
  "suite": {
17
41
  "properties": {
18
- "format": { "const": "agentskills.io/evals" },
19
- "suite_hash": { "type": "string" }
42
+ "format": {
43
+ "const": "agentskills.io/evals"
44
+ },
45
+ "suite_hash": {
46
+ "type": "string"
47
+ }
20
48
  }
21
49
  },
22
50
  "run": {
23
51
  "properties": {
24
- "surface": { "enum": ["api", "claude-cli", "openai-api", "openai-cli"] },
25
- "transcripts": { "enum": ["retained-local", "hashes-only"] }
52
+ "surface": {
53
+ "enum": [
54
+ "api",
55
+ "claude-cli",
56
+ "openai-api",
57
+ "openai-cli"
58
+ ]
59
+ },
60
+ "transcripts": {
61
+ "enum": [
62
+ "retained-local",
63
+ "hashes-only"
64
+ ]
65
+ }
26
66
  }
27
67
  },
28
68
  "comparison": {
29
69
  "properties": {
30
- "baseline_score": { "type": "number" },
31
- "delta": { "type": "number" },
32
- "delta_uncertainty": { "type": "number" }
70
+ "baseline_score": {
71
+ "type": "number"
72
+ },
73
+ "delta": {
74
+ "type": "number"
75
+ },
76
+ "delta_uncertainty": {
77
+ "type": "number"
78
+ }
33
79
  }
34
80
  },
35
81
  "results": {
36
82
  "properties": {
37
83
  "cases": {
38
84
  "items": {
39
- "if": { "not": { "required": ["case_status"], "properties": { "case_status": { "const": "failed_timeout" } } } },
85
+ "if": {
86
+ "not": {
87
+ "required": [
88
+ "case_status"
89
+ ],
90
+ "properties": {
91
+ "case_status": {
92
+ "const": "failed_timeout"
93
+ }
94
+ }
95
+ }
96
+ },
40
97
  "then": {
41
- "required": ["generation_hash", "judge_sample_hashes"],
98
+ "required": [
99
+ "generation_hash",
100
+ "judge_sample_hashes"
101
+ ],
42
102
  "properties": {
43
- "judge": { "properties": { "rubric_hash": { "type": "string" } } }
103
+ "judge": {
104
+ "properties": {
105
+ "rubric_hash": {
106
+ "type": "string"
107
+ }
108
+ }
109
+ }
44
110
  }
45
111
  }
46
112
  }
@@ -54,17 +120,30 @@
54
120
  "properties": {
55
121
  "schema_version": {
56
122
  "type": "string",
57
- "const": "0.3.1"
123
+ "const": "0.4"
58
124
  },
59
125
  "skill": {
60
126
  "type": "object",
61
127
  "additionalProperties": false,
62
- "required": ["name", "version", "content_hash"],
128
+ "required": [
129
+ "name",
130
+ "version",
131
+ "content_hash"
132
+ ],
63
133
  "properties": {
64
- "name": { "type": "string", "minLength": 1 },
65
- "version": { "type": "string", "minLength": 1 },
134
+ "name": {
135
+ "type": "string",
136
+ "minLength": 1
137
+ },
138
+ "version": {
139
+ "type": "string",
140
+ "minLength": 1
141
+ },
66
142
  "content_hash": {
67
- "type": ["string", "null"],
143
+ "type": [
144
+ "string",
145
+ "null"
146
+ ],
68
147
  "description": "sha256 (hex) over SKILL.md + all bundled files in canonical path-sorted order. Null ONLY on an imported (DECLARED) receipt whose skill bytes were never seen; a TESTED receipt must carry the hash (see the TESTED tightening).",
69
148
  "pattern": "^[a-f0-9]{64}$"
70
149
  },
@@ -78,7 +157,11 @@
78
157
  "suite": {
79
158
  "type": "object",
80
159
  "additionalProperties": false,
81
- "required": ["format", "suite_hash", "case_count"],
160
+ "required": [
161
+ "format",
162
+ "suite_hash",
163
+ "case_count"
164
+ ],
82
165
  "properties": {
83
166
  "format": {
84
167
  "type": "string",
@@ -86,32 +169,62 @@
86
169
  "description": "Suite format. Driftproof-run (TESTED) receipts are always 'agentskills.io/evals' (see the TESTED tightening); an imported receipt names the source tool's format (e.g. 'skillgrade/eval.yaml')."
87
170
  },
88
171
  "suite_hash": {
89
- "type": ["string", "null"],
172
+ "type": [
173
+ "string",
174
+ "null"
175
+ ],
90
176
  "description": "sha256 (hex) over the canonicalized normalized case list. Null ONLY on an imported (DECLARED) receipt whose suite bytes were never seen.",
91
177
  "pattern": "^[a-f0-9]{64}$"
92
178
  },
93
- "case_count": { "type": "integer", "minimum": 0 }
179
+ "case_count": {
180
+ "type": "integer",
181
+ "minimum": 0
182
+ }
94
183
  }
95
184
  },
96
185
  "run": {
97
186
  "type": "object",
98
187
  "additionalProperties": false,
99
- "required": ["model_id", "provider", "surface", "runner_version", "date_utc", "judge", "registry", "transcripts"],
188
+ "required": [
189
+ "model_id",
190
+ "provider",
191
+ "surface",
192
+ "runner_version",
193
+ "date_utc",
194
+ "judge",
195
+ "registry",
196
+ "transcripts"
197
+ ],
100
198
  "properties": {
101
- "model_id": { "type": "string", "minLength": 1 },
199
+ "model_id": {
200
+ "type": "string",
201
+ "minLength": 1
202
+ },
102
203
  "model_release_date": {
103
204
  "description": "ISO date (YYYY-MM-DD) of the model release if known, else null.",
104
- "type": ["string", "null"],
205
+ "type": [
206
+ "string",
207
+ "null"
208
+ ],
105
209
  "pattern": "^\\d{4}-\\d{2}-\\d{2}$"
106
210
  },
107
211
  "provider": {
108
212
  "type": "string",
109
213
  "description": "v0.3.1. The two-axis provider the target model ran on (registry `provider`, else inferred from the id).",
110
- "enum": ["anthropic", "openai"]
214
+ "enum": [
215
+ "anthropic",
216
+ "openai"
217
+ ]
111
218
  },
112
219
  "surface": {
113
220
  "type": "string",
114
- "enum": ["api", "claude-cli", "openai-api", "openai-cli", "external"],
221
+ "enum": [
222
+ "api",
223
+ "claude-cli",
224
+ "openai-api",
225
+ "openai-cli",
226
+ "external"
227
+ ],
115
228
  "description": "'external' = the run happened on another tool's harness and was imported (never valid on a TESTED receipt)."
116
229
  },
117
230
  "source": {
@@ -126,14 +239,20 @@
126
239
  "status": {
127
240
  "type": "string",
128
241
  "description": "v0.3.1 (optional; default 'complete'). 'incomplete' when >=1 case persistently failed (e.g. failed_timeout) and was EXCLUDED from aggregates. A drift/durability report must not compute a verdict from an incomplete receipt.",
129
- "enum": ["complete", "incomplete"]
242
+ "enum": [
243
+ "complete",
244
+ "incomplete"
245
+ ]
130
246
  },
131
247
  "failed_case_count": {
132
248
  "type": "integer",
133
249
  "minimum": 0,
134
250
  "description": "v0.3.1 (optional). Number of cases marked failed_timeout (excluded from aggregates)."
135
251
  },
136
- "runner_version": { "type": "string", "minLength": 1 },
252
+ "runner_version": {
253
+ "type": "string",
254
+ "minLength": 1
255
+ },
137
256
  "date_utc": {
138
257
  "type": "string",
139
258
  "description": "ISO 8601 UTC timestamp of when the run finished.",
@@ -142,29 +261,104 @@
142
261
  "registry": {
143
262
  "type": "string",
144
263
  "description": "Whether model_id resolved in the model registry (config/models.json). 'unregistered' means the run still executed but the model was unknown, so cost estimates used the conservative default price.",
145
- "enum": ["registered", "unregistered"]
264
+ "enum": [
265
+ "registered",
266
+ "unregistered"
267
+ ]
146
268
  },
147
269
  "transcripts": {
148
270
  "type": "string",
149
271
  "description": "Transcript retention for this run. 'hashes-only' = only the sha256 hashes in results.cases are kept (the default). 'retained-local' = the raw generations + judge outputs were also written to transcripts/<receipt-id>/ (gitignored by default). 'none' = nothing retained, not even hashes — ONLY honest on an imported (DECLARED) receipt.",
150
- "enum": ["retained-local", "hashes-only", "none"]
272
+ "enum": [
273
+ "retained-local",
274
+ "hashes-only",
275
+ "none"
276
+ ]
151
277
  },
152
278
  "judge": {
153
279
  "type": "object",
154
280
  "additionalProperties": false,
155
281
  "description": "Judge sampling settings for this run.",
156
- "required": ["samples", "temperature", "sampling"],
282
+ "required": [
283
+ "samples",
284
+ "temperature",
285
+ "sampling"
286
+ ],
157
287
  "properties": {
158
- "samples": { "type": "integer", "minimum": 1, "description": "Judge samples taken per case." },
288
+ "samples": {
289
+ "type": "integer",
290
+ "minimum": 1,
291
+ "description": "Judge samples taken per case."
292
+ },
159
293
  "temperature": {
160
- "type": ["number", "null"],
294
+ "type": [
295
+ "number",
296
+ "null"
297
+ ],
161
298
  "description": "Judge temperature when the surface allows setting it (api -> 0), else null (cli -> surface-controlled)."
162
299
  },
163
300
  "sampling": {
164
301
  "type": "string",
165
302
  "description": "How sampling params were controlled, e.g. 'api-temperature-0' or 'surface-controlled'."
166
303
  },
167
- "surface": { "type": "string", "enum": ["api", "claude-cli", "openai-api", "openai-cli", "external"] }
304
+ "surface": {
305
+ "type": "string",
306
+ "enum": [
307
+ "api",
308
+ "claude-cli",
309
+ "openai-api",
310
+ "openai-cli",
311
+ "external"
312
+ ]
313
+ }
314
+ }
315
+ },
316
+ "pricing_snapshot": {
317
+ "type": "object",
318
+ "additionalProperties": false,
319
+ "description": "v0.4: registry prices FROZEN at run time. Every derived dollar figure in `economics` is computed from this snapshot and never from the live registry, so the receipt stays reproducible when registry prices later change.",
320
+ "required": [
321
+ "frozen_at",
322
+ "source",
323
+ "currency",
324
+ "models"
325
+ ],
326
+ "properties": {
327
+ "frozen_at": {
328
+ "type": "string"
329
+ },
330
+ "source": {
331
+ "type": "string"
332
+ },
333
+ "currency": {
334
+ "type": "string"
335
+ },
336
+ "note": {
337
+ "type": "string"
338
+ },
339
+ "models": {
340
+ "type": "object",
341
+ "additionalProperties": {
342
+ "type": "object",
343
+ "additionalProperties": false,
344
+ "required": [
345
+ "input_per_mtok",
346
+ "output_per_mtok",
347
+ "registered"
348
+ ],
349
+ "properties": {
350
+ "input_per_mtok": {
351
+ "type": "number"
352
+ },
353
+ "output_per_mtok": {
354
+ "type": "number"
355
+ },
356
+ "registered": {
357
+ "type": "boolean"
358
+ }
359
+ }
360
+ }
361
+ }
168
362
  }
169
363
  }
170
364
  }
@@ -172,41 +366,105 @@
172
366
  "results": {
173
367
  "type": "object",
174
368
  "additionalProperties": false,
175
- "required": ["cases", "aggregates"],
369
+ "required": [
370
+ "cases",
371
+ "aggregates"
372
+ ],
176
373
  "properties": {
177
374
  "cases": {
178
375
  "type": "array",
179
376
  "items": {
180
377
  "type": "object",
181
378
  "additionalProperties": false,
182
- "required": ["id", "mode"],
379
+ "required": [
380
+ "id",
381
+ "mode"
382
+ ],
183
383
  "allOf": [
184
384
  {
185
385
  "description": "A completed case carries the full sampled band + hashes; a failed_timeout case is recorded WITHOUT fabricated samples (it is excluded from aggregates).",
186
- "if": { "required": ["case_status"], "properties": { "case_status": { "const": "failed_timeout" } } },
187
- "then": { "required": ["id", "mode", "case_status"] },
188
- "else": { "required": ["outcome", "score", "mean", "stddev", "samples", "judge"] }
386
+ "if": {
387
+ "required": [
388
+ "case_status"
389
+ ],
390
+ "properties": {
391
+ "case_status": {
392
+ "const": "failed_timeout"
393
+ }
394
+ }
395
+ },
396
+ "then": {
397
+ "required": [
398
+ "id",
399
+ "mode",
400
+ "case_status"
401
+ ]
402
+ },
403
+ "else": {
404
+ "required": [
405
+ "outcome",
406
+ "score",
407
+ "mean",
408
+ "stddev",
409
+ "samples",
410
+ "judge"
411
+ ]
412
+ }
189
413
  }
190
414
  ],
191
415
  "properties": {
192
- "id": { "type": "string", "minLength": 1 },
193
- "mode": { "type": "string", "enum": ["with_skill", "baseline"] },
416
+ "id": {
417
+ "type": "string",
418
+ "minLength": 1
419
+ },
420
+ "mode": {
421
+ "type": "string",
422
+ "enum": [
423
+ "with_skill",
424
+ "baseline"
425
+ ]
426
+ },
194
427
  "case_status": {
195
428
  "type": "string",
196
429
  "description": "v0.3.1 (optional; default 'ok'). 'failed_timeout' = the case's model/judge call persistently timed out after retries; the case is recorded but EXCLUDED from aggregates/verdicts — no samples/hashes are fabricated for it.",
197
- "enum": ["ok", "failed_timeout"]
430
+ "enum": [
431
+ "ok",
432
+ "failed_timeout"
433
+ ]
198
434
  },
199
435
  "outcome": {
200
436
  "type": "string",
201
437
  "description": "borderline = the threshold lies within mean +/- stddev.",
202
- "enum": ["pass", "fail", "borderline", "score"]
438
+ "enum": [
439
+ "pass",
440
+ "fail",
441
+ "borderline",
442
+ "score"
443
+ ]
444
+ },
445
+ "score": {
446
+ "type": "number",
447
+ "minimum": 0,
448
+ "maximum": 1,
449
+ "description": "Alias of mean, kept for v0.1 readers."
450
+ },
451
+ "mean": {
452
+ "type": "number",
453
+ "minimum": 0,
454
+ "maximum": 1
455
+ },
456
+ "stddev": {
457
+ "type": "number",
458
+ "minimum": 0,
459
+ "description": "Sample stddev of the judge samples (raw band half-width)."
203
460
  },
204
- "score": { "type": "number", "minimum": 0, "maximum": 1, "description": "Alias of mean, kept for v0.1 readers." },
205
- "mean": { "type": "number", "minimum": 0, "maximum": 1 },
206
- "stddev": { "type": "number", "minimum": 0, "description": "Sample stddev of the judge samples (raw band half-width)." },
207
461
  "samples": {
208
462
  "type": "array",
209
- "items": { "type": "number", "minimum": 0, "maximum": 1 },
463
+ "items": {
464
+ "type": "number",
465
+ "minimum": 0,
466
+ "maximum": 1
467
+ },
210
468
  "minItems": 1
211
469
  },
212
470
  "generation_hash": {
@@ -217,37 +475,157 @@
217
475
  "judge_sample_hashes": {
218
476
  "type": "array",
219
477
  "description": "sha256 (hex) of each raw judge output, one per judge sample. Same length as `samples`.",
220
- "items": { "type": "string", "pattern": "^[a-f0-9]{64}$" },
478
+ "items": {
479
+ "type": "string",
480
+ "pattern": "^[a-f0-9]{64}$"
481
+ },
221
482
  "minItems": 1
222
483
  },
223
- "threshold": { "type": ["number", "null"], "minimum": 0, "maximum": 1 },
224
- "reason": { "type": "string" },
484
+ "threshold": {
485
+ "type": [
486
+ "number",
487
+ "null"
488
+ ],
489
+ "minimum": 0,
490
+ "maximum": 1
491
+ },
492
+ "reason": {
493
+ "type": "string"
494
+ },
225
495
  "checks": {
226
496
  "type": "array",
227
497
  "description": "v0.3.1 (optional). Deterministic post-check results for this (case, mode): structural/regex assertions run on the model output ALONGSIDE the judge. Supplementary evidence reported as a separate column — NOT folded into the outcome/band verdict.",
228
498
  "items": {
229
499
  "type": "object",
230
500
  "additionalProperties": false,
231
- "required": ["name", "kind", "pass"],
501
+ "required": [
502
+ "name",
503
+ "kind",
504
+ "pass"
505
+ ],
232
506
  "properties": {
233
- "name": { "type": "string", "minLength": 1 },
234
- "kind": { "type": "string", "enum": ["regex", "contains", "not_contains", "min_length"] },
235
- "pass": { "type": "boolean" }
507
+ "name": {
508
+ "type": "string",
509
+ "minLength": 1
510
+ },
511
+ "kind": {
512
+ "type": "string",
513
+ "enum": [
514
+ "regex",
515
+ "contains",
516
+ "not_contains",
517
+ "min_length"
518
+ ]
519
+ },
520
+ "pass": {
521
+ "type": "boolean"
522
+ }
236
523
  }
237
524
  }
238
525
  },
239
526
  "judge": {
240
527
  "type": "object",
241
528
  "additionalProperties": false,
242
- "required": ["model_id", "rubric_hash"],
529
+ "required": [
530
+ "model_id",
531
+ "rubric_hash"
532
+ ],
243
533
  "properties": {
244
- "model_id": { "type": "string", "minLength": 1 },
534
+ "model_id": {
535
+ "type": "string",
536
+ "minLength": 1
537
+ },
245
538
  "rubric_hash": {
246
- "type": ["string", "null"],
539
+ "type": [
540
+ "string",
541
+ "null"
542
+ ],
247
543
  "description": "Null ONLY on an imported (DECLARED) receipt whose rubric bytes were never seen.",
248
544
  "pattern": "^[a-f0-9]{64}$"
249
545
  }
250
546
  }
547
+ },
548
+ "usage": {
549
+ "type": "object",
550
+ "additionalProperties": false,
551
+ "description": "v0.4: usage of the GENERATION call for this (case, mode) row. Normalized usage for ONE call. input_tokens is the TOTAL input presented to the model INCLUDING any cached portion (the surfaces disagree about this natively; see lib/usage.js). cached_tokens is the portion served from cache, null when the surface does not report it. output_tokens includes reasoning/thinking tokens where the surface bundles them. wall_ms is measured by the runner around the successful attempt, so it means the same thing on every surface. A field the surface did not report is null — never 0.",
552
+ "required": [
553
+ "input_tokens",
554
+ "output_tokens",
555
+ "cached_tokens",
556
+ "wall_ms"
557
+ ],
558
+ "properties": {
559
+ "input_tokens": {
560
+ "type": [
561
+ "integer",
562
+ "null"
563
+ ],
564
+ "minimum": 0
565
+ },
566
+ "output_tokens": {
567
+ "type": [
568
+ "integer",
569
+ "null"
570
+ ],
571
+ "minimum": 0
572
+ },
573
+ "cached_tokens": {
574
+ "type": [
575
+ "integer",
576
+ "null"
577
+ ],
578
+ "minimum": 0
579
+ },
580
+ "wall_ms": {
581
+ "type": [
582
+ "integer",
583
+ "null"
584
+ ],
585
+ "minimum": 0
586
+ }
587
+ }
588
+ },
589
+ "judge_usage": {
590
+ "type": "object",
591
+ "additionalProperties": false,
592
+ "description": "v0.4: SUM of usage over the N judge calls that graded this row. Measurement overhead imposed by the harness, NOT a cost of running the skill — excluded from every field in `economics` by construction.",
593
+ "required": [
594
+ "input_tokens",
595
+ "output_tokens",
596
+ "cached_tokens",
597
+ "wall_ms"
598
+ ],
599
+ "properties": {
600
+ "input_tokens": {
601
+ "type": [
602
+ "integer",
603
+ "null"
604
+ ],
605
+ "minimum": 0
606
+ },
607
+ "output_tokens": {
608
+ "type": [
609
+ "integer",
610
+ "null"
611
+ ],
612
+ "minimum": 0
613
+ },
614
+ "cached_tokens": {
615
+ "type": [
616
+ "integer",
617
+ "null"
618
+ ],
619
+ "minimum": 0
620
+ },
621
+ "wall_ms": {
622
+ "type": [
623
+ "integer",
624
+ "null"
625
+ ],
626
+ "minimum": 0
627
+ }
628
+ }
251
629
  }
252
630
  }
253
631
  }
@@ -255,10 +633,17 @@
255
633
  "aggregates": {
256
634
  "type": "object",
257
635
  "additionalProperties": false,
258
- "required": ["with_skill", "baseline"],
636
+ "required": [
637
+ "with_skill",
638
+ "baseline"
639
+ ],
259
640
  "properties": {
260
- "with_skill": { "$ref": "#/$defs/modeAggregate" },
261
- "baseline": { "$ref": "#/$defs/modeAggregate" }
641
+ "with_skill": {
642
+ "$ref": "#/$defs/modeAggregate"
643
+ },
644
+ "baseline": {
645
+ "$ref": "#/$defs/modeAggregate"
646
+ }
262
647
  }
263
648
  }
264
649
  }
@@ -266,23 +651,41 @@
266
651
  "comparison": {
267
652
  "type": "object",
268
653
  "additionalProperties": false,
269
- "required": ["with_skill_score", "baseline_score", "delta", "delta_uncertainty"],
654
+ "required": [
655
+ "with_skill_score",
656
+ "baseline_score",
657
+ "delta",
658
+ "delta_uncertainty"
659
+ ],
270
660
  "properties": {
271
- "with_skill_score": { "type": "number", "minimum": 0, "maximum": 1 },
661
+ "with_skill_score": {
662
+ "type": "number",
663
+ "minimum": 0,
664
+ "maximum": 1
665
+ },
272
666
  "baseline_score": {
273
- "type": ["number", "null"],
667
+ "type": [
668
+ "number",
669
+ "null"
670
+ ],
274
671
  "minimum": 0,
275
672
  "maximum": 1,
276
673
  "description": "Null ONLY on an imported (DECLARED) receipt from a tool with no baseline mode — never a fabricated 0."
277
674
  },
278
675
  "delta": {
279
- "type": ["number", "null"],
676
+ "type": [
677
+ "number",
678
+ "null"
679
+ ],
280
680
  "minimum": -1,
281
681
  "maximum": 1,
282
682
  "description": "Null when baseline_score is null (no baseline mode was run)."
283
683
  },
284
684
  "delta_uncertainty": {
285
- "type": ["number", "null"],
685
+ "type": [
686
+ "number",
687
+ "null"
688
+ ],
286
689
  "minimum": 0,
287
690
  "description": "Combined uncertainty of the delta (quadrature sum of the two aggregate bands). Null when delta is null."
288
691
  }
@@ -291,7 +694,11 @@
291
694
  "verification_level": {
292
695
  "type": "string",
293
696
  "description": "Community verification lattice. FORMAL is reserved/unimplemented in v0.3.",
294
- "enum": ["UNVERIFIED", "DECLARED", "TESTED"]
697
+ "enum": [
698
+ "UNVERIFIED",
699
+ "DECLARED",
700
+ "TESTED"
701
+ ]
295
702
  },
296
703
  "editorial_reviews": {
297
704
  "type": "array",
@@ -299,11 +706,24 @@
299
706
  "items": {
300
707
  "type": "object",
301
708
  "additionalProperties": false,
302
- "required": ["url", "source", "date"],
709
+ "required": [
710
+ "url",
711
+ "source",
712
+ "date"
713
+ ],
303
714
  "properties": {
304
- "url": { "type": "string", "minLength": 1 },
305
- "source": { "type": "string", "minLength": 1 },
306
- "date": { "type": "string", "pattern": "^\\d{4}-\\d{2}-\\d{2}$" }
715
+ "url": {
716
+ "type": "string",
717
+ "minLength": 1
718
+ },
719
+ "source": {
720
+ "type": "string",
721
+ "minLength": 1
722
+ },
723
+ "date": {
724
+ "type": "string",
725
+ "pattern": "^\\d{4}-\\d{2}-\\d{2}$"
726
+ }
307
727
  }
308
728
  }
309
729
  },
@@ -311,18 +731,225 @@
311
731
  "type": "string",
312
732
  "description": "sha256 (hex) of the canonical receipt JSON with this field omitted.",
313
733
  "pattern": "^[a-f0-9]{64}$"
734
+ },
735
+ "economics": {
736
+ "type": "object",
737
+ "additionalProperties": false,
738
+ "description": "v0.4 derived economics. Computed from the per-case `usage` at `run.pricing_snapshot` prices. The three value axes (accuracy lift, cost, latency) live separately here and in the reports: there is deliberately NO composite value score, because collapsing axes with different units and different error bars would produce a number no reader could trace to evidence.",
739
+ "required": [
740
+ "basis",
741
+ "with_skill",
742
+ "baseline",
743
+ "judge_excluded"
744
+ ],
745
+ "properties": {
746
+ "basis": {
747
+ "enum": [
748
+ "metered",
749
+ "metered-equivalent"
750
+ ],
751
+ "description": "metered = real spend on an api surface; metered-equivalent = what the same tokens would have cost on the metered API (subscription CLI surfaces, where actual spend is $0)."
752
+ },
753
+ "surface": {
754
+ "type": "string"
755
+ },
756
+ "with_skill": {
757
+ "type": "object",
758
+ "additionalProperties": false,
759
+ "properties": {
760
+ "call_count": {
761
+ "type": "integer",
762
+ "minimum": 0
763
+ },
764
+ "mean_input_tokens": {
765
+ "type": [
766
+ "number",
767
+ "null"
768
+ ]
769
+ },
770
+ "mean_output_tokens": {
771
+ "type": [
772
+ "number",
773
+ "null"
774
+ ]
775
+ },
776
+ "mean_cost_usd_per_call": {
777
+ "type": [
778
+ "number",
779
+ "null"
780
+ ]
781
+ },
782
+ "median_wall_ms": {
783
+ "type": [
784
+ "number",
785
+ "null"
786
+ ]
787
+ },
788
+ "wall_ms_p25": {
789
+ "type": [
790
+ "number",
791
+ "null"
792
+ ]
793
+ },
794
+ "wall_ms_p75": {
795
+ "type": [
796
+ "number",
797
+ "null"
798
+ ]
799
+ },
800
+ "wall_ms_iqr": {
801
+ "type": [
802
+ "number",
803
+ "null"
804
+ ]
805
+ }
806
+ }
807
+ },
808
+ "baseline": {
809
+ "type": "object",
810
+ "additionalProperties": false,
811
+ "properties": {
812
+ "call_count": {
813
+ "type": "integer",
814
+ "minimum": 0
815
+ },
816
+ "mean_input_tokens": {
817
+ "type": [
818
+ "number",
819
+ "null"
820
+ ]
821
+ },
822
+ "mean_output_tokens": {
823
+ "type": [
824
+ "number",
825
+ "null"
826
+ ]
827
+ },
828
+ "mean_cost_usd_per_call": {
829
+ "type": [
830
+ "number",
831
+ "null"
832
+ ]
833
+ },
834
+ "median_wall_ms": {
835
+ "type": [
836
+ "number",
837
+ "null"
838
+ ]
839
+ },
840
+ "wall_ms_p25": {
841
+ "type": [
842
+ "number",
843
+ "null"
844
+ ]
845
+ },
846
+ "wall_ms_p75": {
847
+ "type": [
848
+ "number",
849
+ "null"
850
+ ]
851
+ },
852
+ "wall_ms_iqr": {
853
+ "type": [
854
+ "number",
855
+ "null"
856
+ ]
857
+ }
858
+ }
859
+ },
860
+ "skill_incremental_cost_usd_per_call": {
861
+ "type": [
862
+ "number",
863
+ "null"
864
+ ]
865
+ },
866
+ "skill_incremental_cost_usd_per_1k_calls": {
867
+ "type": [
868
+ "number",
869
+ "null"
870
+ ]
871
+ },
872
+ "output_tokens_delta": {
873
+ "type": [
874
+ "number",
875
+ "null"
876
+ ]
877
+ },
878
+ "median_wall_ms_delta": {
879
+ "type": [
880
+ "number",
881
+ "null"
882
+ ]
883
+ },
884
+ "judge_excluded": {
885
+ "const": true,
886
+ "description": "Structural guarantee: judge usage never enters any figure in this block. A receipt cannot claim otherwise."
887
+ },
888
+ "judge_overhead": {
889
+ "type": "object",
890
+ "additionalProperties": false,
891
+ "properties": {
892
+ "note": {
893
+ "type": "string"
894
+ },
895
+ "total_cost_usd": {
896
+ "type": [
897
+ "number",
898
+ "null"
899
+ ]
900
+ },
901
+ "case_rows_measured": {
902
+ "type": "integer",
903
+ "minimum": 0
904
+ }
905
+ }
906
+ },
907
+ "notes": {
908
+ "type": "object",
909
+ "additionalProperties": false,
910
+ "properties": {
911
+ "absolute_cost": {
912
+ "type": "string"
913
+ },
914
+ "cache_pricing": {
915
+ "type": "string"
916
+ },
917
+ "latency": {
918
+ "type": "string"
919
+ }
920
+ }
921
+ }
922
+ }
314
923
  }
315
924
  },
316
925
  "$defs": {
317
926
  "modeAggregate": {
318
927
  "type": "object",
319
928
  "additionalProperties": false,
320
- "required": ["case_count", "pass_count", "mean_score", "stddev"],
929
+ "required": [
930
+ "case_count",
931
+ "pass_count",
932
+ "mean_score",
933
+ "stddev"
934
+ ],
321
935
  "properties": {
322
- "case_count": { "type": "integer", "minimum": 0 },
323
- "pass_count": { "type": "integer", "minimum": 0 },
324
- "borderline_count": { "type": "integer", "minimum": 0 },
325
- "mean_score": { "type": "number", "minimum": 0, "maximum": 1 },
936
+ "case_count": {
937
+ "type": "integer",
938
+ "minimum": 0
939
+ },
940
+ "pass_count": {
941
+ "type": "integer",
942
+ "minimum": 0
943
+ },
944
+ "borderline_count": {
945
+ "type": "integer",
946
+ "minimum": 0
947
+ },
948
+ "mean_score": {
949
+ "type": "number",
950
+ "minimum": 0,
951
+ "maximum": 1
952
+ },
326
953
  "stddev": {
327
954
  "type": "number",
328
955
  "minimum": 0,