driftproof 0.6.0 → 0.7.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,961 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://driftproofhq.com/spec/receipt.v0.4.schema.json",
4
+ "title": "driftproof receipt",
5
+ "description": "A hash-verified, dated record of running one agent skill's eval suite with and without the skill on one model version, with sampled judge scores and confidence bands. Receipt spec v0.4 — additive over v0.3.1: adds per-case, per-arm generation `usage` (input/output/cached tokens + measured wall_ms) captured from the surfaces that report it, a separate per-case `judge_usage` (measurement overhead, EXCLUDED from every skill-value figure by construction — `economics.judge_excluded` is const true), a run-level `run.pricing_snapshot` freezing the registry prices the derived dollar figures were computed from (so a receipt keeps its meaning when prices later change), and a derived `economics` block (per-arm mean cost/call, skill incremental cost per call and per 1k calls, output-length delta, median wall_ms with IQR). The three value axes — accuracy lift, cost, latency — are recorded separately and NEVER combined into a composite score. All v0.3.1 semantics are unchanged and every prior receipt still validates against its own frozen schema (v0.1, v0.2, v0.3, v0.3.1). The TESTED tightening (see allOf) is unchanged: the interop relaxations remain available only below TESTED.",
6
+ "type": "object",
7
+ "additionalProperties": false,
8
+ "required": [
9
+ "schema_version",
10
+ "skill",
11
+ "suite",
12
+ "run",
13
+ "results",
14
+ "comparison",
15
+ "verification_level",
16
+ "receipt_hash"
17
+ ],
18
+ "allOf": [
19
+ {
20
+ "description": "TESTED tightening — the interop relaxations (null hashes, external surface, null comparison, transcripts 'none') are ONLY available to receipts below TESTED. A TESTED receipt must carry the full evidence chain, exactly as before the interop revision.",
21
+ "if": {
22
+ "required": [
23
+ "verification_level"
24
+ ],
25
+ "properties": {
26
+ "verification_level": {
27
+ "const": "TESTED"
28
+ }
29
+ }
30
+ },
31
+ "then": {
32
+ "properties": {
33
+ "skill": {
34
+ "properties": {
35
+ "content_hash": {
36
+ "type": "string"
37
+ }
38
+ }
39
+ },
40
+ "suite": {
41
+ "properties": {
42
+ "format": {
43
+ "const": "agentskills.io/evals"
44
+ },
45
+ "suite_hash": {
46
+ "type": "string"
47
+ }
48
+ }
49
+ },
50
+ "run": {
51
+ "properties": {
52
+ "surface": {
53
+ "enum": [
54
+ "api",
55
+ "claude-cli",
56
+ "openai-api",
57
+ "openai-cli"
58
+ ]
59
+ },
60
+ "transcripts": {
61
+ "enum": [
62
+ "retained-local",
63
+ "hashes-only"
64
+ ]
65
+ }
66
+ }
67
+ },
68
+ "comparison": {
69
+ "properties": {
70
+ "baseline_score": {
71
+ "type": "number"
72
+ },
73
+ "delta": {
74
+ "type": "number"
75
+ },
76
+ "delta_uncertainty": {
77
+ "type": "number"
78
+ }
79
+ }
80
+ },
81
+ "results": {
82
+ "properties": {
83
+ "cases": {
84
+ "items": {
85
+ "if": {
86
+ "not": {
87
+ "required": [
88
+ "case_status"
89
+ ],
90
+ "properties": {
91
+ "case_status": {
92
+ "const": "failed_timeout"
93
+ }
94
+ }
95
+ }
96
+ },
97
+ "then": {
98
+ "required": [
99
+ "generation_hash",
100
+ "judge_sample_hashes"
101
+ ],
102
+ "properties": {
103
+ "judge": {
104
+ "properties": {
105
+ "rubric_hash": {
106
+ "type": "string"
107
+ }
108
+ }
109
+ }
110
+ }
111
+ }
112
+ }
113
+ }
114
+ }
115
+ }
116
+ }
117
+ }
118
+ }
119
+ ],
120
+ "properties": {
121
+ "schema_version": {
122
+ "type": "string",
123
+ "const": "0.4"
124
+ },
125
+ "skill": {
126
+ "type": "object",
127
+ "additionalProperties": false,
128
+ "required": [
129
+ "name",
130
+ "version",
131
+ "content_hash"
132
+ ],
133
+ "properties": {
134
+ "name": {
135
+ "type": "string",
136
+ "minLength": 1
137
+ },
138
+ "version": {
139
+ "type": "string",
140
+ "minLength": 1
141
+ },
142
+ "content_hash": {
143
+ "type": [
144
+ "string",
145
+ "null"
146
+ ],
147
+ "description": "sha256 (hex) over SKILL.md + all bundled files in canonical path-sorted order. Null ONLY on an imported (DECLARED) receipt whose skill bytes were never seen; a TESTED receipt must carry the hash (see the TESTED tightening).",
148
+ "pattern": "^[a-f0-9]{64}$"
149
+ },
150
+ "tokens": {
151
+ "type": "integer",
152
+ "minimum": 0,
153
+ "description": "v0.3.1 (optional). Estimated token size of the skill's SKILL.md (a coarse chars/4 proxy, not a model tokenizer), used for the value-per-token axis (delta per 1k skill tokens). See docs/methodology.html."
154
+ }
155
+ }
156
+ },
157
+ "suite": {
158
+ "type": "object",
159
+ "additionalProperties": false,
160
+ "required": [
161
+ "format",
162
+ "suite_hash",
163
+ "case_count"
164
+ ],
165
+ "properties": {
166
+ "format": {
167
+ "type": "string",
168
+ "minLength": 1,
169
+ "description": "Suite format. Driftproof-run (TESTED) receipts are always 'agentskills.io/evals' (see the TESTED tightening); an imported receipt names the source tool's format (e.g. 'skillgrade/eval.yaml')."
170
+ },
171
+ "suite_hash": {
172
+ "type": [
173
+ "string",
174
+ "null"
175
+ ],
176
+ "description": "sha256 (hex) over the canonicalized normalized case list. Null ONLY on an imported (DECLARED) receipt whose suite bytes were never seen.",
177
+ "pattern": "^[a-f0-9]{64}$"
178
+ },
179
+ "case_count": {
180
+ "type": "integer",
181
+ "minimum": 0
182
+ }
183
+ }
184
+ },
185
+ "run": {
186
+ "type": "object",
187
+ "additionalProperties": false,
188
+ "required": [
189
+ "model_id",
190
+ "provider",
191
+ "surface",
192
+ "runner_version",
193
+ "date_utc",
194
+ "judge",
195
+ "registry",
196
+ "transcripts"
197
+ ],
198
+ "properties": {
199
+ "model_id": {
200
+ "type": "string",
201
+ "minLength": 1
202
+ },
203
+ "model_release_date": {
204
+ "description": "ISO date (YYYY-MM-DD) of the model release if known, else null.",
205
+ "type": [
206
+ "string",
207
+ "null"
208
+ ],
209
+ "pattern": "^\\d{4}-\\d{2}-\\d{2}$"
210
+ },
211
+ "provider": {
212
+ "type": "string",
213
+ "description": "v0.3.1. The two-axis provider the target model ran on (registry `provider`, else inferred from the id).",
214
+ "enum": [
215
+ "anthropic",
216
+ "openai"
217
+ ]
218
+ },
219
+ "surface": {
220
+ "type": "string",
221
+ "enum": [
222
+ "api",
223
+ "claude-cli",
224
+ "openai-api",
225
+ "openai-cli",
226
+ "external"
227
+ ],
228
+ "description": "'external' = the run happened on another tool's harness and was imported (never valid on a TESTED receipt)."
229
+ },
230
+ "source": {
231
+ "type": "string",
232
+ "minLength": 1,
233
+ "description": "Interop-additive (optional). Provenance of a converted receipt, e.g. 'imported/agent-skills-eval' or 'imported/skillgrade'. Absent on receipts Driftproof ran itself."
234
+ },
235
+ "surface_overhead_note": {
236
+ "type": "string",
237
+ "description": "v0.3.1 (optional). Present on the openai/cli surface: states the fixed Codex base-instruction preamble (~12–15k input tokens per call) that the harness prepends and does not control."
238
+ },
239
+ "status": {
240
+ "type": "string",
241
+ "description": "v0.3.1 (optional; default 'complete'). 'incomplete' when >=1 case persistently failed (e.g. failed_timeout) and was EXCLUDED from aggregates. A drift/durability report must not compute a verdict from an incomplete receipt.",
242
+ "enum": [
243
+ "complete",
244
+ "incomplete"
245
+ ]
246
+ },
247
+ "failed_case_count": {
248
+ "type": "integer",
249
+ "minimum": 0,
250
+ "description": "v0.3.1 (optional). Number of cases marked failed_timeout (excluded from aggregates)."
251
+ },
252
+ "runner_version": {
253
+ "type": "string",
254
+ "minLength": 1
255
+ },
256
+ "date_utc": {
257
+ "type": "string",
258
+ "description": "ISO 8601 UTC timestamp of when the run finished.",
259
+ "pattern": "^\\d{4}-\\d{2}-\\d{2}T\\d{2}:\\d{2}:\\d{2}"
260
+ },
261
+ "registry": {
262
+ "type": "string",
263
+ "description": "Whether model_id resolved in the model registry (config/models.json). 'unregistered' means the run still executed but the model was unknown, so cost estimates used the conservative default price.",
264
+ "enum": [
265
+ "registered",
266
+ "unregistered"
267
+ ]
268
+ },
269
+ "transcripts": {
270
+ "type": "string",
271
+ "description": "Transcript retention for this run. 'hashes-only' = only the sha256 hashes in results.cases are kept (the default). 'retained-local' = the raw generations + judge outputs were also written to transcripts/<receipt-id>/ (gitignored by default). 'none' = nothing retained, not even hashes — ONLY honest on an imported (DECLARED) receipt.",
272
+ "enum": [
273
+ "retained-local",
274
+ "hashes-only",
275
+ "none"
276
+ ]
277
+ },
278
+ "judge": {
279
+ "type": "object",
280
+ "additionalProperties": false,
281
+ "description": "Judge sampling settings for this run.",
282
+ "required": [
283
+ "samples",
284
+ "temperature",
285
+ "sampling"
286
+ ],
287
+ "properties": {
288
+ "samples": {
289
+ "type": "integer",
290
+ "minimum": 1,
291
+ "description": "Judge samples taken per case."
292
+ },
293
+ "temperature": {
294
+ "type": [
295
+ "number",
296
+ "null"
297
+ ],
298
+ "description": "Judge temperature when the surface allows setting it (api -> 0), else null (cli -> surface-controlled)."
299
+ },
300
+ "sampling": {
301
+ "type": "string",
302
+ "description": "How sampling params were controlled, e.g. 'api-temperature-0' or 'surface-controlled'."
303
+ },
304
+ "surface": {
305
+ "type": "string",
306
+ "enum": [
307
+ "api",
308
+ "claude-cli",
309
+ "openai-api",
310
+ "openai-cli",
311
+ "external"
312
+ ]
313
+ }
314
+ }
315
+ },
316
+ "pricing_snapshot": {
317
+ "type": "object",
318
+ "additionalProperties": false,
319
+ "description": "v0.4: registry prices FROZEN at run time. Every derived dollar figure in `economics` is computed from this snapshot and never from the live registry, so the receipt stays reproducible when registry prices later change.",
320
+ "required": [
321
+ "frozen_at",
322
+ "source",
323
+ "currency",
324
+ "models"
325
+ ],
326
+ "properties": {
327
+ "frozen_at": {
328
+ "type": "string"
329
+ },
330
+ "source": {
331
+ "type": "string"
332
+ },
333
+ "currency": {
334
+ "type": "string"
335
+ },
336
+ "note": {
337
+ "type": "string"
338
+ },
339
+ "models": {
340
+ "type": "object",
341
+ "additionalProperties": {
342
+ "type": "object",
343
+ "additionalProperties": false,
344
+ "required": [
345
+ "input_per_mtok",
346
+ "output_per_mtok",
347
+ "registered"
348
+ ],
349
+ "properties": {
350
+ "input_per_mtok": {
351
+ "type": "number"
352
+ },
353
+ "output_per_mtok": {
354
+ "type": "number"
355
+ },
356
+ "registered": {
357
+ "type": "boolean"
358
+ }
359
+ }
360
+ }
361
+ }
362
+ }
363
+ }
364
+ }
365
+ },
366
+ "results": {
367
+ "type": "object",
368
+ "additionalProperties": false,
369
+ "required": [
370
+ "cases",
371
+ "aggregates"
372
+ ],
373
+ "properties": {
374
+ "cases": {
375
+ "type": "array",
376
+ "items": {
377
+ "type": "object",
378
+ "additionalProperties": false,
379
+ "required": [
380
+ "id",
381
+ "mode"
382
+ ],
383
+ "allOf": [
384
+ {
385
+ "description": "A completed case carries the full sampled band + hashes; a failed_timeout case is recorded WITHOUT fabricated samples (it is excluded from aggregates).",
386
+ "if": {
387
+ "required": [
388
+ "case_status"
389
+ ],
390
+ "properties": {
391
+ "case_status": {
392
+ "const": "failed_timeout"
393
+ }
394
+ }
395
+ },
396
+ "then": {
397
+ "required": [
398
+ "id",
399
+ "mode",
400
+ "case_status"
401
+ ]
402
+ },
403
+ "else": {
404
+ "required": [
405
+ "outcome",
406
+ "score",
407
+ "mean",
408
+ "stddev",
409
+ "samples",
410
+ "judge"
411
+ ]
412
+ }
413
+ }
414
+ ],
415
+ "properties": {
416
+ "id": {
417
+ "type": "string",
418
+ "minLength": 1
419
+ },
420
+ "mode": {
421
+ "type": "string",
422
+ "enum": [
423
+ "with_skill",
424
+ "baseline"
425
+ ]
426
+ },
427
+ "case_status": {
428
+ "type": "string",
429
+ "description": "v0.3.1 (optional; default 'ok'). 'failed_timeout' = the case's model/judge call persistently timed out after retries; the case is recorded but EXCLUDED from aggregates/verdicts — no samples/hashes are fabricated for it.",
430
+ "enum": [
431
+ "ok",
432
+ "failed_timeout"
433
+ ]
434
+ },
435
+ "outcome": {
436
+ "type": "string",
437
+ "description": "borderline = the threshold lies within mean +/- stddev.",
438
+ "enum": [
439
+ "pass",
440
+ "fail",
441
+ "borderline",
442
+ "score"
443
+ ]
444
+ },
445
+ "score": {
446
+ "type": "number",
447
+ "minimum": 0,
448
+ "maximum": 1,
449
+ "description": "Alias of mean, kept for v0.1 readers."
450
+ },
451
+ "mean": {
452
+ "type": "number",
453
+ "minimum": 0,
454
+ "maximum": 1
455
+ },
456
+ "stddev": {
457
+ "type": "number",
458
+ "minimum": 0,
459
+ "description": "Sample stddev of the judge samples (raw band half-width)."
460
+ },
461
+ "samples": {
462
+ "type": "array",
463
+ "items": {
464
+ "type": "number",
465
+ "minimum": 0,
466
+ "maximum": 1
467
+ },
468
+ "minItems": 1
469
+ },
470
+ "generation_hash": {
471
+ "type": "string",
472
+ "description": "sha256 (hex) of the raw model generation that was judged for this (case, mode).",
473
+ "pattern": "^[a-f0-9]{64}$"
474
+ },
475
+ "judge_sample_hashes": {
476
+ "type": "array",
477
+ "description": "sha256 (hex) of each raw judge output, one per judge sample. Same length as `samples`.",
478
+ "items": {
479
+ "type": "string",
480
+ "pattern": "^[a-f0-9]{64}$"
481
+ },
482
+ "minItems": 1
483
+ },
484
+ "threshold": {
485
+ "type": [
486
+ "number",
487
+ "null"
488
+ ],
489
+ "minimum": 0,
490
+ "maximum": 1
491
+ },
492
+ "reason": {
493
+ "type": "string"
494
+ },
495
+ "checks": {
496
+ "type": "array",
497
+ "description": "v0.3.1 (optional). Deterministic post-check results for this (case, mode): structural/regex assertions run on the model output ALONGSIDE the judge. Supplementary evidence reported as a separate column — NOT folded into the outcome/band verdict.",
498
+ "items": {
499
+ "type": "object",
500
+ "additionalProperties": false,
501
+ "required": [
502
+ "name",
503
+ "kind",
504
+ "pass"
505
+ ],
506
+ "properties": {
507
+ "name": {
508
+ "type": "string",
509
+ "minLength": 1
510
+ },
511
+ "kind": {
512
+ "type": "string",
513
+ "enum": [
514
+ "regex",
515
+ "contains",
516
+ "not_contains",
517
+ "min_length"
518
+ ]
519
+ },
520
+ "pass": {
521
+ "type": "boolean"
522
+ }
523
+ }
524
+ }
525
+ },
526
+ "judge": {
527
+ "type": "object",
528
+ "additionalProperties": false,
529
+ "required": [
530
+ "model_id",
531
+ "rubric_hash"
532
+ ],
533
+ "properties": {
534
+ "model_id": {
535
+ "type": "string",
536
+ "minLength": 1
537
+ },
538
+ "rubric_hash": {
539
+ "type": [
540
+ "string",
541
+ "null"
542
+ ],
543
+ "description": "Null ONLY on an imported (DECLARED) receipt whose rubric bytes were never seen.",
544
+ "pattern": "^[a-f0-9]{64}$"
545
+ }
546
+ }
547
+ },
548
+ "usage": {
549
+ "type": "object",
550
+ "additionalProperties": false,
551
+ "description": "v0.4: usage of the GENERATION call for this (case, mode) row. Normalized usage for ONE call. input_tokens is the TOTAL input presented to the model INCLUDING any cached portion (the surfaces disagree about this natively; see lib/usage.js). cached_tokens is the portion served from cache, null when the surface does not report it. output_tokens includes reasoning/thinking tokens where the surface bundles them. wall_ms is measured by the runner around the successful attempt, so it means the same thing on every surface. A field the surface did not report is null — never 0.",
552
+ "required": [
553
+ "input_tokens",
554
+ "output_tokens",
555
+ "cached_tokens",
556
+ "wall_ms"
557
+ ],
558
+ "properties": {
559
+ "input_tokens": {
560
+ "type": [
561
+ "integer",
562
+ "null"
563
+ ],
564
+ "minimum": 0
565
+ },
566
+ "output_tokens": {
567
+ "type": [
568
+ "integer",
569
+ "null"
570
+ ],
571
+ "minimum": 0
572
+ },
573
+ "cached_tokens": {
574
+ "type": [
575
+ "integer",
576
+ "null"
577
+ ],
578
+ "minimum": 0
579
+ },
580
+ "wall_ms": {
581
+ "type": [
582
+ "integer",
583
+ "null"
584
+ ],
585
+ "minimum": 0
586
+ }
587
+ }
588
+ },
589
+ "judge_usage": {
590
+ "type": "object",
591
+ "additionalProperties": false,
592
+ "description": "v0.4: SUM of usage over the N judge calls that graded this row. Measurement overhead imposed by the harness, NOT a cost of running the skill — excluded from every field in `economics` by construction.",
593
+ "required": [
594
+ "input_tokens",
595
+ "output_tokens",
596
+ "cached_tokens",
597
+ "wall_ms"
598
+ ],
599
+ "properties": {
600
+ "input_tokens": {
601
+ "type": [
602
+ "integer",
603
+ "null"
604
+ ],
605
+ "minimum": 0
606
+ },
607
+ "output_tokens": {
608
+ "type": [
609
+ "integer",
610
+ "null"
611
+ ],
612
+ "minimum": 0
613
+ },
614
+ "cached_tokens": {
615
+ "type": [
616
+ "integer",
617
+ "null"
618
+ ],
619
+ "minimum": 0
620
+ },
621
+ "wall_ms": {
622
+ "type": [
623
+ "integer",
624
+ "null"
625
+ ],
626
+ "minimum": 0
627
+ }
628
+ }
629
+ }
630
+ }
631
+ }
632
+ },
633
+ "aggregates": {
634
+ "type": "object",
635
+ "additionalProperties": false,
636
+ "required": [
637
+ "with_skill",
638
+ "baseline"
639
+ ],
640
+ "properties": {
641
+ "with_skill": {
642
+ "$ref": "#/$defs/modeAggregate"
643
+ },
644
+ "baseline": {
645
+ "$ref": "#/$defs/modeAggregate"
646
+ }
647
+ }
648
+ }
649
+ }
650
+ },
651
+ "comparison": {
652
+ "type": "object",
653
+ "additionalProperties": false,
654
+ "required": [
655
+ "with_skill_score",
656
+ "baseline_score",
657
+ "delta",
658
+ "delta_uncertainty"
659
+ ],
660
+ "properties": {
661
+ "with_skill_score": {
662
+ "type": "number",
663
+ "minimum": 0,
664
+ "maximum": 1
665
+ },
666
+ "baseline_score": {
667
+ "type": [
668
+ "number",
669
+ "null"
670
+ ],
671
+ "minimum": 0,
672
+ "maximum": 1,
673
+ "description": "Null ONLY on an imported (DECLARED) receipt from a tool with no baseline mode — never a fabricated 0."
674
+ },
675
+ "delta": {
676
+ "type": [
677
+ "number",
678
+ "null"
679
+ ],
680
+ "minimum": -1,
681
+ "maximum": 1,
682
+ "description": "Null when baseline_score is null (no baseline mode was run)."
683
+ },
684
+ "delta_uncertainty": {
685
+ "type": [
686
+ "number",
687
+ "null"
688
+ ],
689
+ "minimum": 0,
690
+ "description": "Combined uncertainty of the delta (quadrature sum of the two aggregate bands). Null when delta is null."
691
+ }
692
+ }
693
+ },
694
+ "verification_level": {
695
+ "type": "string",
696
+ "description": "Community verification lattice. FORMAL is reserved/unimplemented in v0.3.",
697
+ "enum": [
698
+ "UNVERIFIED",
699
+ "DECLARED",
700
+ "TESTED"
701
+ ]
702
+ },
703
+ "editorial_reviews": {
704
+ "type": "array",
705
+ "description": "Optional pointers to external one-shot editorial reviews of this skill (context only; not verification evidence).",
706
+ "items": {
707
+ "type": "object",
708
+ "additionalProperties": false,
709
+ "required": [
710
+ "url",
711
+ "source",
712
+ "date"
713
+ ],
714
+ "properties": {
715
+ "url": {
716
+ "type": "string",
717
+ "minLength": 1
718
+ },
719
+ "source": {
720
+ "type": "string",
721
+ "minLength": 1
722
+ },
723
+ "date": {
724
+ "type": "string",
725
+ "pattern": "^\\d{4}-\\d{2}-\\d{2}$"
726
+ }
727
+ }
728
+ }
729
+ },
730
+ "receipt_hash": {
731
+ "type": "string",
732
+ "description": "sha256 (hex) of the canonical receipt JSON with this field omitted.",
733
+ "pattern": "^[a-f0-9]{64}$"
734
+ },
735
+ "economics": {
736
+ "type": "object",
737
+ "additionalProperties": false,
738
+ "description": "v0.4 derived economics. Computed from the per-case `usage` at `run.pricing_snapshot` prices. The three value axes (accuracy lift, cost, latency) live separately here and in the reports: there is deliberately NO composite value score, because collapsing axes with different units and different error bars would produce a number no reader could trace to evidence.",
739
+ "required": [
740
+ "basis",
741
+ "with_skill",
742
+ "baseline",
743
+ "judge_excluded"
744
+ ],
745
+ "properties": {
746
+ "basis": {
747
+ "enum": [
748
+ "metered",
749
+ "metered-equivalent"
750
+ ],
751
+ "description": "metered = real spend on an api surface; metered-equivalent = what the same tokens would have cost on the metered API (subscription CLI surfaces, where actual spend is $0)."
752
+ },
753
+ "surface": {
754
+ "type": "string"
755
+ },
756
+ "with_skill": {
757
+ "type": "object",
758
+ "additionalProperties": false,
759
+ "properties": {
760
+ "call_count": {
761
+ "type": "integer",
762
+ "minimum": 0
763
+ },
764
+ "mean_input_tokens": {
765
+ "type": [
766
+ "number",
767
+ "null"
768
+ ]
769
+ },
770
+ "mean_output_tokens": {
771
+ "type": [
772
+ "number",
773
+ "null"
774
+ ]
775
+ },
776
+ "mean_cost_usd_per_call": {
777
+ "type": [
778
+ "number",
779
+ "null"
780
+ ]
781
+ },
782
+ "median_wall_ms": {
783
+ "type": [
784
+ "number",
785
+ "null"
786
+ ]
787
+ },
788
+ "wall_ms_p25": {
789
+ "type": [
790
+ "number",
791
+ "null"
792
+ ]
793
+ },
794
+ "wall_ms_p75": {
795
+ "type": [
796
+ "number",
797
+ "null"
798
+ ]
799
+ },
800
+ "wall_ms_iqr": {
801
+ "type": [
802
+ "number",
803
+ "null"
804
+ ]
805
+ }
806
+ }
807
+ },
808
+ "baseline": {
809
+ "type": "object",
810
+ "additionalProperties": false,
811
+ "properties": {
812
+ "call_count": {
813
+ "type": "integer",
814
+ "minimum": 0
815
+ },
816
+ "mean_input_tokens": {
817
+ "type": [
818
+ "number",
819
+ "null"
820
+ ]
821
+ },
822
+ "mean_output_tokens": {
823
+ "type": [
824
+ "number",
825
+ "null"
826
+ ]
827
+ },
828
+ "mean_cost_usd_per_call": {
829
+ "type": [
830
+ "number",
831
+ "null"
832
+ ]
833
+ },
834
+ "median_wall_ms": {
835
+ "type": [
836
+ "number",
837
+ "null"
838
+ ]
839
+ },
840
+ "wall_ms_p25": {
841
+ "type": [
842
+ "number",
843
+ "null"
844
+ ]
845
+ },
846
+ "wall_ms_p75": {
847
+ "type": [
848
+ "number",
849
+ "null"
850
+ ]
851
+ },
852
+ "wall_ms_iqr": {
853
+ "type": [
854
+ "number",
855
+ "null"
856
+ ]
857
+ }
858
+ }
859
+ },
860
+ "skill_incremental_cost_usd_per_call": {
861
+ "type": [
862
+ "number",
863
+ "null"
864
+ ]
865
+ },
866
+ "skill_incremental_cost_usd_per_1k_calls": {
867
+ "type": [
868
+ "number",
869
+ "null"
870
+ ]
871
+ },
872
+ "output_tokens_delta": {
873
+ "type": [
874
+ "number",
875
+ "null"
876
+ ]
877
+ },
878
+ "median_wall_ms_delta": {
879
+ "type": [
880
+ "number",
881
+ "null"
882
+ ]
883
+ },
884
+ "judge_excluded": {
885
+ "const": true,
886
+ "description": "Structural guarantee: judge usage never enters any figure in this block. A receipt cannot claim otherwise."
887
+ },
888
+ "judge_overhead": {
889
+ "type": "object",
890
+ "additionalProperties": false,
891
+ "properties": {
892
+ "note": {
893
+ "type": "string"
894
+ },
895
+ "total_cost_usd": {
896
+ "type": [
897
+ "number",
898
+ "null"
899
+ ]
900
+ },
901
+ "case_rows_measured": {
902
+ "type": "integer",
903
+ "minimum": 0
904
+ }
905
+ }
906
+ },
907
+ "notes": {
908
+ "type": "object",
909
+ "additionalProperties": false,
910
+ "properties": {
911
+ "absolute_cost": {
912
+ "type": "string"
913
+ },
914
+ "cache_pricing": {
915
+ "type": "string"
916
+ },
917
+ "latency": {
918
+ "type": "string"
919
+ }
920
+ }
921
+ }
922
+ }
923
+ }
924
+ },
925
+ "$defs": {
926
+ "modeAggregate": {
927
+ "type": "object",
928
+ "additionalProperties": false,
929
+ "required": [
930
+ "case_count",
931
+ "pass_count",
932
+ "mean_score",
933
+ "stddev"
934
+ ],
935
+ "properties": {
936
+ "case_count": {
937
+ "type": "integer",
938
+ "minimum": 0
939
+ },
940
+ "pass_count": {
941
+ "type": "integer",
942
+ "minimum": 0
943
+ },
944
+ "borderline_count": {
945
+ "type": "integer",
946
+ "minimum": 0
947
+ },
948
+ "mean_score": {
949
+ "type": "number",
950
+ "minimum": 0,
951
+ "maximum": 1
952
+ },
953
+ "stddev": {
954
+ "type": "number",
955
+ "minimum": 0,
956
+ "description": "Suite dispersion: stddev of the per-case means across the suite. A reported summary stat; the drift headline is driven by per-case band-overlap verdicts, not this band."
957
+ }
958
+ }
959
+ }
960
+ }
961
+ }