driftproof 0.11.2 → 0.11.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1563 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://driftproofhq.com/spec/receipt.v0.7.schema.json",
4
+ "title": "driftproof receipt",
5
+ "description": "A hash-verified, dated record of running one agent skill eval suite with and without the skill on one model version, with the GENERATION sampled n times per arm and the judge sampled k times inside each draw. Receipt spec v0.7: over v0.6, the fields are unchanged and the verdict a reader derives gains UNDERPOWERED: a case the band rule does not separate whose spreads and measured draws cannot resolve a shift of the effect floor (spec 035, spec/RECEIPT.md § What v0.7 adds). Over v0.5, as v0.6 said, the receipt says what answered it: `run.answered_by` (kind model|stub|external, whether the surface echoed a model id and which, and which spawn path produced it); `run.surface` and `run.judge.surface` may read `stub` (nothing answered; the text was canned) and a TESTED receipt requires answered_by.kind model; `run.judge.model_id` and `run.judge.prompt_template_hash` say which judge ran; every draw records its `stop_reason` and whether it was `truncated` (a truncated draw is unmeasured, never judged) and the case counts `n_truncated`; `case_status` gains `failed_unmeasured` (every draw of the arm was unmeasured for a reason that is not a timeout), recorded without fabricated samples or hashes exactly as a timeout is; `results.aggregates.band_rule` states the formula the aggregate band is derived by; an aggregate `stddev` is null only when its arm has fewer than two cases and `comparison.delta_uncertainty` is null only beside `delta_uncertainty_unavailable`. Every prior receipt still validates against its own frozen schema (v0.1, v0.2, v0.3, v0.3.1, v0.4, v0.5, v0.6).",
6
+ "type": "object",
7
+ "additionalProperties": false,
8
+ "required": [
9
+ "schema_version",
10
+ "skill",
11
+ "suite",
12
+ "run",
13
+ "results",
14
+ "comparison",
15
+ "verification_level",
16
+ "receipt_hash"
17
+ ],
18
+ "allOf": [
19
+ {
20
+ "description": "TESTED tightening — the interop relaxations (null hashes, external surface, null comparison, transcripts 'none') are ONLY available to receipts below TESTED. A TESTED receipt must carry the full evidence chain, exactly as before the interop revision. v0.6: a TESTED receipt was answered by a model (answered_by.kind model), its judge template is recorded, and its comparison is numeric except where delta_uncertainty_unavailable names why a band could not form.",
21
+ "if": {
22
+ "required": [
23
+ "verification_level"
24
+ ],
25
+ "properties": {
26
+ "verification_level": {
27
+ "const": "TESTED"
28
+ }
29
+ }
30
+ },
31
+ "then": {
32
+ "properties": {
33
+ "skill": {
34
+ "properties": {
35
+ "content_hash": {
36
+ "type": "string"
37
+ }
38
+ }
39
+ },
40
+ "suite": {
41
+ "properties": {
42
+ "format": {
43
+ "const": "agentskills.io/evals"
44
+ },
45
+ "suite_hash": {
46
+ "type": "string"
47
+ }
48
+ }
49
+ },
50
+ "run": {
51
+ "properties": {
52
+ "surface": {
53
+ "enum": [
54
+ "api",
55
+ "claude-cli",
56
+ "openai-api",
57
+ "openai-cli"
58
+ ]
59
+ },
60
+ "transcripts": {
61
+ "enum": [
62
+ "retained-local",
63
+ "hashes-only"
64
+ ]
65
+ },
66
+ "answered_by": {
67
+ "properties": {
68
+ "kind": {
69
+ "const": "model"
70
+ }
71
+ }
72
+ },
73
+ "judge": {
74
+ "properties": {
75
+ "prompt_template_hash": {
76
+ "type": "string"
77
+ }
78
+ }
79
+ }
80
+ }
81
+ },
82
+ "comparison": {
83
+ "allOf": [
84
+ {
85
+ "if": {
86
+ "not": {
87
+ "required": [
88
+ "delta_uncertainty_unavailable"
89
+ ]
90
+ }
91
+ },
92
+ "then": {
93
+ "properties": {
94
+ "baseline_score": {
95
+ "type": "number"
96
+ },
97
+ "delta": {
98
+ "type": "number"
99
+ },
100
+ "delta_uncertainty": {
101
+ "type": "number"
102
+ }
103
+ }
104
+ }
105
+ },
106
+ {
107
+ "if": {
108
+ "required": [
109
+ "delta_uncertainty_unavailable"
110
+ ],
111
+ "properties": {
112
+ "delta_uncertainty_unavailable": {
113
+ "const": "single_case"
114
+ }
115
+ }
116
+ },
117
+ "then": {
118
+ "properties": {
119
+ "with_skill_score": {
120
+ "type": "number"
121
+ },
122
+ "baseline_score": {
123
+ "type": "number"
124
+ },
125
+ "delta": {
126
+ "type": "number"
127
+ }
128
+ }
129
+ }
130
+ }
131
+ ]
132
+ },
133
+ "results": {
134
+ "properties": {
135
+ "cases": {
136
+ "items": {
137
+ "if": {
138
+ "anyOf": [
139
+ {
140
+ "not": {
141
+ "required": [
142
+ "case_status"
143
+ ]
144
+ }
145
+ },
146
+ {
147
+ "properties": {
148
+ "case_status": {
149
+ "const": "ok"
150
+ }
151
+ }
152
+ }
153
+ ]
154
+ },
155
+ "then": {
156
+ "required": [
157
+ "generation_hash",
158
+ "judge_sample_hashes"
159
+ ],
160
+ "properties": {
161
+ "judge": {
162
+ "properties": {
163
+ "rubric_hash": {
164
+ "type": "string"
165
+ }
166
+ }
167
+ }
168
+ }
169
+ }
170
+ }
171
+ }
172
+ }
173
+ }
174
+ }
175
+ }
176
+ },
177
+ {
178
+ "$comment": "driftproof/generation-sampled-declared",
179
+ "description": "F-014-F, first direction: a receipt carrying a draw set or a variance ratio must declare the capability. Without this a third-party emitter validates as v0.5 while carrying none of what v0.5 exists to add.",
180
+ "if": {
181
+ "required": [
182
+ "results"
183
+ ],
184
+ "properties": {
185
+ "results": {
186
+ "required": [
187
+ "cases"
188
+ ],
189
+ "properties": {
190
+ "cases": {
191
+ "contains": {
192
+ "required": [
193
+ "generation"
194
+ ]
195
+ }
196
+ }
197
+ }
198
+ }
199
+ }
200
+ },
201
+ "then": {
202
+ "required": [
203
+ "generation_sampled"
204
+ ],
205
+ "properties": {
206
+ "generation_sampled": {
207
+ "const": true
208
+ }
209
+ }
210
+ }
211
+ },
212
+ {
213
+ "$comment": "driftproof/generation-sampled-honest",
214
+ "description": "F-014-F, second direction: a receipt that declares the capability must carry it. A flag assertable by a receipt carrying nothing would be a second way to claim a capability falsely, and would close the exposure in one direction only.",
215
+ "if": {
216
+ "required": [
217
+ "generation_sampled"
218
+ ],
219
+ "properties": {
220
+ "generation_sampled": {
221
+ "const": true
222
+ }
223
+ }
224
+ },
225
+ "then": {
226
+ "required": [
227
+ "results"
228
+ ],
229
+ "properties": {
230
+ "results": {
231
+ "required": [
232
+ "cases"
233
+ ],
234
+ "properties": {
235
+ "cases": {
236
+ "contains": {
237
+ "required": [
238
+ "generation"
239
+ ]
240
+ }
241
+ }
242
+ }
243
+ }
244
+ }
245
+ }
246
+ },
247
+ {
248
+ "$comment": "driftproof/generation-sampled-canary",
249
+ "description": "F-015-A, the other half of F-014-F: a receipt declaring the capability must also carry the per-suite canary. F-014-F named BOTH results.cases[].generation and suite.canary; spec 015 bound the first and left this one, so a receipt could still claim v0.5 conformance while carrying only half of what v0.5 adds. A receipt that declares generation_sampled has by construction run a suite of ours, so it has a canary to record; omitting it is the same false claim in the other half. Legacy is untouched: a v0.4-and-earlier receipt is governed by its own frozen schema, and a v0.5 receipt that ran no generation sampling declares nothing and is unaffected.",
250
+ "if": {
251
+ "required": [
252
+ "generation_sampled"
253
+ ],
254
+ "properties": {
255
+ "generation_sampled": {
256
+ "const": true
257
+ }
258
+ }
259
+ },
260
+ "then": {
261
+ "required": [
262
+ "suite"
263
+ ],
264
+ "properties": {
265
+ "suite": {
266
+ "required": [
267
+ "canary"
268
+ ]
269
+ }
270
+ }
271
+ }
272
+ }
273
+ ],
274
+ "properties": {
275
+ "schema_version": {
276
+ "type": "string",
277
+ "const": "0.7"
278
+ },
279
+ "skill": {
280
+ "type": "object",
281
+ "additionalProperties": false,
282
+ "required": [
283
+ "name",
284
+ "version",
285
+ "content_hash"
286
+ ],
287
+ "properties": {
288
+ "name": {
289
+ "type": "string",
290
+ "minLength": 1
291
+ },
292
+ "version": {
293
+ "type": "string",
294
+ "minLength": 1
295
+ },
296
+ "content_hash": {
297
+ "type": [
298
+ "string",
299
+ "null"
300
+ ],
301
+ "description": "sha256 (hex) over SKILL.md + all bundled files in canonical path-sorted order. Null ONLY on an imported (DECLARED) receipt whose skill bytes were never seen; a TESTED receipt must carry the hash (see the TESTED tightening).",
302
+ "pattern": "^[a-f0-9]{64}$"
303
+ },
304
+ "tokens": {
305
+ "type": "integer",
306
+ "minimum": 0,
307
+ "description": "v0.3.1 (optional). Estimated token size of the skill's SKILL.md (a coarse chars/4 proxy, not a model tokenizer), used for the value-per-token axis (delta per 1k skill tokens). See docs/methodology.html."
308
+ }
309
+ }
310
+ },
311
+ "suite": {
312
+ "type": "object",
313
+ "additionalProperties": false,
314
+ "required": [
315
+ "format",
316
+ "suite_hash",
317
+ "case_count"
318
+ ],
319
+ "properties": {
320
+ "format": {
321
+ "type": "string",
322
+ "minLength": 1,
323
+ "description": "Suite format. Driftproof-run (TESTED) receipts are always 'agentskills.io/evals' (see the TESTED tightening); an imported receipt names the source tool's format (e.g. 'skillgrade/eval.yaml')."
324
+ },
325
+ "suite_hash": {
326
+ "type": [
327
+ "string",
328
+ "null"
329
+ ],
330
+ "description": "sha256 (hex) over the canonicalized normalized case list. Null ONLY on an imported (DECLARED) receipt whose suite bytes were never seen.",
331
+ "pattern": "^[a-f0-9]{64}$"
332
+ },
333
+ "case_count": {
334
+ "type": "integer",
335
+ "minimum": 0
336
+ },
337
+ "canary": {
338
+ "type": "string",
339
+ "pattern": "^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$",
340
+ "description": "Per-suite canary GUID: a leaked suite is detectable in a corpus. Detection aid, not a control."
341
+ }
342
+ }
343
+ },
344
+ "run": {
345
+ "type": "object",
346
+ "additionalProperties": false,
347
+ "required": [
348
+ "model_id",
349
+ "provider",
350
+ "surface",
351
+ "runner_version",
352
+ "date_utc",
353
+ "judge",
354
+ "registry",
355
+ "transcripts",
356
+ "answered_by"
357
+ ],
358
+ "properties": {
359
+ "model_id": {
360
+ "type": "string",
361
+ "minLength": 1
362
+ },
363
+ "model_release_date": {
364
+ "description": "ISO date (YYYY-MM-DD) of the model release if known, else null.",
365
+ "type": [
366
+ "string",
367
+ "null"
368
+ ],
369
+ "pattern": "^\\d{4}-\\d{2}-\\d{2}$"
370
+ },
371
+ "provider": {
372
+ "type": "string",
373
+ "description": "v0.3.1. The two-axis provider the target model ran on (registry `provider`, else inferred from the id).",
374
+ "enum": [
375
+ "anthropic",
376
+ "openai"
377
+ ]
378
+ },
379
+ "surface": {
380
+ "type": "string",
381
+ "enum": [
382
+ "api",
383
+ "claude-cli",
384
+ "openai-api",
385
+ "openai-cli",
386
+ "external",
387
+ "stub"
388
+ ],
389
+ "description": "'external' = the run happened on another tool's harness and was imported (never valid on a TESTED receipt). 'stub' = nothing answered (DRIFTPROOF_STUB=1); the text was canned. Never valid on a TESTED receipt."
390
+ },
391
+ "source": {
392
+ "type": "string",
393
+ "minLength": 1,
394
+ "description": "Interop-additive (optional). Provenance of a converted receipt, e.g. 'imported/agent-skills-eval' or 'imported/skillgrade'. Absent on receipts Driftproof ran itself."
395
+ },
396
+ "surface_overhead_note": {
397
+ "type": "string",
398
+ "description": "v0.3.1 (optional). Present on the openai/cli surface: states the fixed Codex base-instruction preamble (~12–15k input tokens per call) that the harness prepends and does not control."
399
+ },
400
+ "status": {
401
+ "type": "string",
402
+ "description": "v0.3.1 (optional; default 'complete'). 'incomplete' when >=1 case persistently failed (e.g. failed_timeout) and was EXCLUDED from aggregates. A drift/durability report must not compute a verdict from an incomplete receipt.",
403
+ "enum": [
404
+ "complete",
405
+ "incomplete"
406
+ ]
407
+ },
408
+ "failed_case_count": {
409
+ "type": "integer",
410
+ "minimum": 0,
411
+ "description": "v0.3.1 (optional; redefined in v0.5). The number of CASES excluded from the aggregates: exactly the length of results.aggregates.excluded_cases, computed from that list so the two cannot disagree. Exclusion is PAIRWISE since v0.5 — a case with any unmeasured arm leaves BOTH arms together and is counted ONCE here, however many of its arms failed."
412
+ },
413
+ "runner_version": {
414
+ "type": "string",
415
+ "minLength": 1
416
+ },
417
+ "date_utc": {
418
+ "type": "string",
419
+ "description": "ISO 8601 UTC timestamp of when the run finished.",
420
+ "pattern": "^\\d{4}-\\d{2}-\\d{2}T\\d{2}:\\d{2}:\\d{2}"
421
+ },
422
+ "registry": {
423
+ "type": "string",
424
+ "description": "Whether model_id resolved in the model registry (config/models.json). 'unregistered' means the run still executed but the model was unknown, so cost estimates used the conservative default price.",
425
+ "enum": [
426
+ "registered",
427
+ "unregistered"
428
+ ]
429
+ },
430
+ "transcripts": {
431
+ "type": "string",
432
+ "description": "Transcript retention for this run. 'hashes-only' = only the sha256 hashes in results.cases are kept (the default). 'retained-local' = the raw generations + judge outputs were also written to transcripts/<receipt-id>/ (gitignored by default). 'none' = nothing retained, not even hashes — ONLY honest on an imported (DECLARED) receipt.",
433
+ "enum": [
434
+ "retained-local",
435
+ "hashes-only",
436
+ "none"
437
+ ]
438
+ },
439
+ "judge": {
440
+ "type": "object",
441
+ "additionalProperties": false,
442
+ "description": "Judge sampling settings for this run.",
443
+ "required": [
444
+ "samples",
445
+ "temperature",
446
+ "sampling",
447
+ "model_id",
448
+ "prompt_template_hash"
449
+ ],
450
+ "properties": {
451
+ "samples": {
452
+ "type": "integer",
453
+ "minimum": 1,
454
+ "description": "Judge samples taken per case."
455
+ },
456
+ "temperature": {
457
+ "type": [
458
+ "number",
459
+ "null"
460
+ ],
461
+ "description": "Judge temperature when the surface allows setting it (api -> 0), else null (cli -> surface-controlled)."
462
+ },
463
+ "sampling": {
464
+ "type": "string",
465
+ "description": "How sampling params were controlled, e.g. 'api-temperature-0' or 'surface-controlled'."
466
+ },
467
+ "surface": {
468
+ "type": "string",
469
+ "enum": [
470
+ "api",
471
+ "claude-cli",
472
+ "openai-api",
473
+ "openai-cli",
474
+ "external",
475
+ "stub"
476
+ ]
477
+ },
478
+ "model_id": {
479
+ "type": "string",
480
+ "minLength": 1,
481
+ "description": "v0.6. The judge model that graded every case of this run; every case's `judge.model_id` equals it on a receipt whose answered_by.kind is model."
482
+ },
483
+ "prompt_template_hash": {
484
+ "type": [
485
+ "string",
486
+ "null"
487
+ ],
488
+ "pattern": "^[a-f0-9]{64}$",
489
+ "description": "v0.6. sha256 over the grading template with its three slots (rubric, task, response) empty: a different template is a different judge. Null ONLY on an imported receipt (answered_by.kind external)."
490
+ }
491
+ }
492
+ },
493
+ "pricing_snapshot": {
494
+ "type": "object",
495
+ "additionalProperties": false,
496
+ "description": "v0.4: registry prices FROZEN at run time. Every derived dollar figure in `economics` is computed from this snapshot and never from the live registry, so the receipt stays reproducible when registry prices later change.",
497
+ "required": [
498
+ "frozen_at",
499
+ "source",
500
+ "currency",
501
+ "models"
502
+ ],
503
+ "properties": {
504
+ "frozen_at": {
505
+ "type": "string"
506
+ },
507
+ "source": {
508
+ "type": "string"
509
+ },
510
+ "currency": {
511
+ "type": "string"
512
+ },
513
+ "note": {
514
+ "type": "string"
515
+ },
516
+ "models": {
517
+ "type": "object",
518
+ "additionalProperties": {
519
+ "type": "object",
520
+ "additionalProperties": false,
521
+ "required": [
522
+ "input_per_mtok",
523
+ "output_per_mtok",
524
+ "registered"
525
+ ],
526
+ "properties": {
527
+ "input_per_mtok": {
528
+ "type": "number"
529
+ },
530
+ "output_per_mtok": {
531
+ "type": "number"
532
+ },
533
+ "registered": {
534
+ "type": "boolean"
535
+ }
536
+ }
537
+ }
538
+ }
539
+ }
540
+ },
541
+ "answered_by": {
542
+ "type": "object",
543
+ "additionalProperties": false,
544
+ "description": "v0.6. What answered the run. `kind` model = a model surface answered; stub = the canned stub answered, nothing was measured; external = another tool's harness answered and the receipt was imported. `attested` = the surface echoed a model id for every draw and it matched the requested canonical id. `reported_model` = the id the surface echoed (null when it echoed nothing); `reported_models` = every id the surface named across the run (null when none). `isolation` = the spawn path: eval-user (the isolated hop), same-user (--trusted-skill), none (the stub, or an api surface that spawns nothing).",
545
+ "required": [
546
+ "kind",
547
+ "attested",
548
+ "reported_model",
549
+ "reported_models",
550
+ "isolation"
551
+ ],
552
+ "properties": {
553
+ "kind": {
554
+ "type": "string",
555
+ "enum": [
556
+ "model",
557
+ "stub",
558
+ "external"
559
+ ]
560
+ },
561
+ "attested": {
562
+ "type": "boolean"
563
+ },
564
+ "reported_model": {
565
+ "type": [
566
+ "string",
567
+ "null"
568
+ ],
569
+ "minLength": 1
570
+ },
571
+ "reported_models": {
572
+ "type": [
573
+ "array",
574
+ "null"
575
+ ],
576
+ "items": {
577
+ "type": "string",
578
+ "minLength": 1
579
+ }
580
+ },
581
+ "isolation": {
582
+ "type": "string",
583
+ "enum": [
584
+ "eval-user",
585
+ "same-user",
586
+ "none"
587
+ ]
588
+ }
589
+ }
590
+ }
591
+ }
592
+ },
593
+ "results": {
594
+ "type": "object",
595
+ "additionalProperties": false,
596
+ "required": [
597
+ "cases",
598
+ "aggregates"
599
+ ],
600
+ "properties": {
601
+ "cases": {
602
+ "type": "array",
603
+ "items": {
604
+ "type": "object",
605
+ "additionalProperties": false,
606
+ "required": [
607
+ "id",
608
+ "mode"
609
+ ],
610
+ "allOf": [
611
+ {
612
+ "description": "A completed case carries the full sampled band + hashes; a failed case (any case_status that is not ok) is recorded WITHOUT fabricated samples (it is excluded from aggregates). v0.6 keys this on case_status not ok rather than on the failed_timeout literal.",
613
+ "if": {
614
+ "anyOf": [
615
+ {
616
+ "not": {
617
+ "required": [
618
+ "case_status"
619
+ ]
620
+ }
621
+ },
622
+ {
623
+ "properties": {
624
+ "case_status": {
625
+ "const": "ok"
626
+ }
627
+ }
628
+ }
629
+ ]
630
+ },
631
+ "then": {
632
+ "required": [
633
+ "outcome",
634
+ "score",
635
+ "mean",
636
+ "stddev",
637
+ "samples",
638
+ "judge"
639
+ ]
640
+ },
641
+ "else": {
642
+ "required": [
643
+ "id",
644
+ "mode",
645
+ "case_status"
646
+ ]
647
+ }
648
+ }
649
+ ],
650
+ "properties": {
651
+ "id": {
652
+ "type": "string",
653
+ "minLength": 1
654
+ },
655
+ "mode": {
656
+ "type": "string",
657
+ "enum": [
658
+ "with_skill",
659
+ "baseline"
660
+ ]
661
+ },
662
+ "case_status": {
663
+ "type": "string",
664
+ "description": "v0.3.1 (optional; default 'ok'). 'failed_timeout' = the case's model/judge call persistently timed out after retries. v0.6: 'failed_unmeasured' = every draw of the arm was unmeasured for a reason that is not a timeout (an empty generation, a judge output with no score, a truncated draw). Either way the case is recorded but EXCLUDED from aggregates/verdicts — no samples/hashes are fabricated for it.",
665
+ "enum": [
666
+ "ok",
667
+ "failed_timeout",
668
+ "failed_unmeasured"
669
+ ]
670
+ },
671
+ "outcome": {
672
+ "type": "string",
673
+ "description": "borderline = the threshold lies within mean +/- stddev.",
674
+ "enum": [
675
+ "pass",
676
+ "fail",
677
+ "borderline",
678
+ "score"
679
+ ]
680
+ },
681
+ "score": {
682
+ "type": "number",
683
+ "minimum": 0,
684
+ "maximum": 1,
685
+ "description": "Alias of mean, kept for v0.1 readers."
686
+ },
687
+ "mean": {
688
+ "type": "number",
689
+ "minimum": 0,
690
+ "maximum": 1
691
+ },
692
+ "stddev": {
693
+ "type": "number",
694
+ "minimum": 0,
695
+ "description": "Sample stddev of the judge samples (raw band half-width)."
696
+ },
697
+ "samples": {
698
+ "type": "array",
699
+ "items": {
700
+ "type": "number",
701
+ "minimum": 0,
702
+ "maximum": 1
703
+ },
704
+ "minItems": 1
705
+ },
706
+ "generation_hash": {
707
+ "type": "string",
708
+ "description": "sha256 (hex) of the raw model generation that was judged for this (case, mode).",
709
+ "pattern": "^[a-f0-9]{64}$"
710
+ },
711
+ "judge_sample_hashes": {
712
+ "type": "array",
713
+ "description": "sha256 (hex) of each raw judge output, one per judge sample. Same length as `samples`.",
714
+ "items": {
715
+ "type": "string",
716
+ "pattern": "^[a-f0-9]{64}$"
717
+ },
718
+ "minItems": 1
719
+ },
720
+ "threshold": {
721
+ "type": [
722
+ "number",
723
+ "null"
724
+ ],
725
+ "minimum": 0,
726
+ "maximum": 1
727
+ },
728
+ "reason": {
729
+ "type": "string"
730
+ },
731
+ "checks": {
732
+ "type": "array",
733
+ "description": "v0.3.1 (optional). Deterministic post-check results for this (case, mode): structural/regex assertions run on the model output ALONGSIDE the judge. Supplementary evidence reported as a separate column — NOT folded into the outcome/band verdict.",
734
+ "items": {
735
+ "type": "object",
736
+ "additionalProperties": false,
737
+ "required": [
738
+ "name",
739
+ "kind",
740
+ "pass"
741
+ ],
742
+ "properties": {
743
+ "name": {
744
+ "type": "string",
745
+ "minLength": 1
746
+ },
747
+ "kind": {
748
+ "type": "string",
749
+ "enum": [
750
+ "regex",
751
+ "contains",
752
+ "not_contains",
753
+ "min_length"
754
+ ]
755
+ },
756
+ "pass": {
757
+ "type": "boolean"
758
+ }
759
+ }
760
+ }
761
+ },
762
+ "judge": {
763
+ "type": "object",
764
+ "additionalProperties": false,
765
+ "required": [
766
+ "model_id",
767
+ "rubric_hash"
768
+ ],
769
+ "properties": {
770
+ "model_id": {
771
+ "type": "string",
772
+ "minLength": 1
773
+ },
774
+ "rubric_hash": {
775
+ "type": [
776
+ "string",
777
+ "null"
778
+ ],
779
+ "description": "Null ONLY on an imported (DECLARED) receipt whose rubric bytes were never seen.",
780
+ "pattern": "^[a-f0-9]{64}$"
781
+ }
782
+ }
783
+ },
784
+ "usage": {
785
+ "type": "object",
786
+ "additionalProperties": false,
787
+ "description": "v0.4: usage of the GENERATION call for this (case, mode) row. Normalized usage for ONE call. input_tokens is the TOTAL input presented to the model INCLUDING any cached portion (the surfaces disagree about this natively; see lib/usage.js). cached_tokens is the portion served from cache, null when the surface does not report it. output_tokens includes reasoning/thinking tokens where the surface bundles them. wall_ms is measured by the runner around the successful attempt, so it means the same thing on every surface. A field the surface did not report is null — never 0.",
788
+ "required": [
789
+ "input_tokens",
790
+ "output_tokens",
791
+ "cached_tokens",
792
+ "wall_ms"
793
+ ],
794
+ "properties": {
795
+ "input_tokens": {
796
+ "type": [
797
+ "integer",
798
+ "null"
799
+ ],
800
+ "minimum": 0
801
+ },
802
+ "output_tokens": {
803
+ "type": [
804
+ "integer",
805
+ "null"
806
+ ],
807
+ "minimum": 0
808
+ },
809
+ "cached_tokens": {
810
+ "type": [
811
+ "integer",
812
+ "null"
813
+ ],
814
+ "minimum": 0
815
+ },
816
+ "wall_ms": {
817
+ "type": [
818
+ "integer",
819
+ "null"
820
+ ],
821
+ "minimum": 0
822
+ }
823
+ }
824
+ },
825
+ "judge_usage": {
826
+ "type": "object",
827
+ "additionalProperties": false,
828
+ "description": "v0.4: SUM of usage over the N judge calls that graded this row. Measurement overhead imposed by the harness, NOT a cost of running the skill — excluded from every field in `economics` by construction.",
829
+ "required": [
830
+ "input_tokens",
831
+ "output_tokens",
832
+ "cached_tokens",
833
+ "wall_ms"
834
+ ],
835
+ "properties": {
836
+ "input_tokens": {
837
+ "type": [
838
+ "integer",
839
+ "null"
840
+ ],
841
+ "minimum": 0
842
+ },
843
+ "output_tokens": {
844
+ "type": [
845
+ "integer",
846
+ "null"
847
+ ],
848
+ "minimum": 0
849
+ },
850
+ "cached_tokens": {
851
+ "type": [
852
+ "integer",
853
+ "null"
854
+ ],
855
+ "minimum": 0
856
+ },
857
+ "wall_ms": {
858
+ "type": [
859
+ "integer",
860
+ "null"
861
+ ],
862
+ "minimum": 0
863
+ }
864
+ }
865
+ },
866
+ "generation": {
867
+ "type": "object",
868
+ "additionalProperties": true,
869
+ "required": [
870
+ "n_drawn",
871
+ "n_measured",
872
+ "n_unmeasured",
873
+ "stopping_reason",
874
+ "draws",
875
+ "n_truncated"
876
+ ],
877
+ "properties": {
878
+ "n_planned": {
879
+ "type": "integer",
880
+ "minimum": 1
881
+ },
882
+ "n_drawn": {
883
+ "type": "integer",
884
+ "minimum": 0
885
+ },
886
+ "n_measured": {
887
+ "type": "integer",
888
+ "minimum": 0
889
+ },
890
+ "n_unmeasured": {
891
+ "type": "integer",
892
+ "minimum": 0
893
+ },
894
+ "stopping_reason": {
895
+ "enum": [
896
+ "min_reached",
897
+ "stabilised",
898
+ "max_reached",
899
+ "unmeasured_exhausted",
900
+ "below_min",
901
+ "escalating"
902
+ ]
903
+ },
904
+ "mean": {
905
+ "type": [
906
+ "number",
907
+ "null"
908
+ ]
909
+ },
910
+ "sd": {
911
+ "type": [
912
+ "number",
913
+ "null"
914
+ ]
915
+ },
916
+ "judge_sd_mean": {
917
+ "type": [
918
+ "number",
919
+ "null"
920
+ ]
921
+ },
922
+ "variance_ratio": {
923
+ "type": [
924
+ "number",
925
+ "null"
926
+ ]
927
+ },
928
+ "draws": {
929
+ "type": "array",
930
+ "items": {
931
+ "type": "object",
932
+ "additionalProperties": true,
933
+ "required": [
934
+ "draw_index",
935
+ "status",
936
+ "stop_reason",
937
+ "truncated"
938
+ ],
939
+ "properties": {
940
+ "draw_index": {
941
+ "type": "integer",
942
+ "minimum": 0
943
+ },
944
+ "status": {
945
+ "enum": [
946
+ "measured",
947
+ "unmeasured"
948
+ ]
949
+ },
950
+ "generation_hash": {
951
+ "type": [
952
+ "string",
953
+ "null"
954
+ ]
955
+ },
956
+ "samples": {
957
+ "type": "array",
958
+ "items": {
959
+ "type": "number"
960
+ }
961
+ },
962
+ "judge_sample_hashes": {
963
+ "type": "array",
964
+ "items": {
965
+ "type": "string"
966
+ }
967
+ },
968
+ "mean": {
969
+ "type": [
970
+ "number",
971
+ "null"
972
+ ]
973
+ },
974
+ "stddev": {
975
+ "type": [
976
+ "number",
977
+ "null"
978
+ ]
979
+ },
980
+ "reason": {
981
+ "type": "string"
982
+ },
983
+ "usage": {
984
+ "type": "object"
985
+ },
986
+ "judge_usage": {
987
+ "type": "object"
988
+ },
989
+ "stop_reason": {
990
+ "type": [
991
+ "string",
992
+ "null"
993
+ ],
994
+ "description": "v0.6. Why the generation stopped, as the surface reported it (e.g. end_turn, max_tokens); null when the surface reports none."
995
+ },
996
+ "truncated": {
997
+ "type": "boolean",
998
+ "description": "v0.6. True when the generation was cut at the output cap. A truncated draw is unmeasured."
999
+ },
1000
+ "reported_model": {
1001
+ "type": [
1002
+ "string",
1003
+ "null"
1004
+ ],
1005
+ "description": "v0.6. The model id the surface echoed for this draw; null when it echoed nothing."
1006
+ }
1007
+ },
1008
+ "allOf": [
1009
+ {
1010
+ "if": {
1011
+ "properties": {
1012
+ "status": {
1013
+ "const": "unmeasured"
1014
+ }
1015
+ },
1016
+ "required": [
1017
+ "status"
1018
+ ]
1019
+ },
1020
+ "then": {
1021
+ "properties": {
1022
+ "mean": {
1023
+ "type": "null"
1024
+ },
1025
+ "stddev": {
1026
+ "type": "null"
1027
+ },
1028
+ "samples": {
1029
+ "maxItems": 0
1030
+ }
1031
+ }
1032
+ }
1033
+ },
1034
+ {
1035
+ "if": {
1036
+ "properties": {
1037
+ "status": {
1038
+ "const": "measured"
1039
+ }
1040
+ },
1041
+ "required": [
1042
+ "status"
1043
+ ]
1044
+ },
1045
+ "then": {
1046
+ "required": [
1047
+ "generation_hash",
1048
+ "samples",
1049
+ "mean",
1050
+ "stddev"
1051
+ ],
1052
+ "properties": {
1053
+ "mean": {
1054
+ "type": "number"
1055
+ },
1056
+ "generation_hash": {
1057
+ "type": "string"
1058
+ }
1059
+ }
1060
+ }
1061
+ }
1062
+ ]
1063
+ }
1064
+ },
1065
+ "variance_ratio_unavailable": {
1066
+ "type": [
1067
+ "string",
1068
+ "null"
1069
+ ],
1070
+ "enum": [
1071
+ "single_judge_sample",
1072
+ "judge_sd_zero",
1073
+ "no_measured_draws",
1074
+ "judge_samples_unknown",
1075
+ null
1076
+ ],
1077
+ "description": "WHICH null the variance_ratio is (F-014-C). `null` when a ratio was formed. `single_judge_sample`: k=1, so the per-draw judge spread is 0 by construction and the ratio is undefined — this is the shape every pre-v0.5 example produced. `judge_sd_zero`: k>=2 and the judge agreed with itself perfectly inside every measured draw. `no_measured_draws`: nothing was measured. `judge_samples_unknown`: the draws carry no sample list, so the count cannot be established and saying which null it is would assert a cause this control cannot reach. Each names what was OBSERVED and none names a cause."
1078
+ },
1079
+ "n_truncated": {
1080
+ "type": "integer",
1081
+ "minimum": 0,
1082
+ "description": "v0.6. Draws whose generation stopped at the output cap (stop_reason max_tokens or the surface's equivalent); each is unmeasured, never judged, and excluded from every statistic."
1083
+ }
1084
+ },
1085
+ "allOf": [
1086
+ {
1087
+ "$comment": "driftproof/variance-ratio-cause",
1088
+ "description": "F-014-C: a null ratio must say WHICH null it is. Bound to the null rather than required outright, so a receipt that formed a ratio carries no reason and nothing in the archive is retroactively incomplete.",
1089
+ "if": {
1090
+ "required": [
1091
+ "variance_ratio"
1092
+ ],
1093
+ "properties": {
1094
+ "variance_ratio": {
1095
+ "type": "null"
1096
+ }
1097
+ }
1098
+ },
1099
+ "then": {
1100
+ "required": [
1101
+ "variance_ratio_unavailable"
1102
+ ],
1103
+ "properties": {
1104
+ "variance_ratio_unavailable": {
1105
+ "type": "string"
1106
+ }
1107
+ }
1108
+ }
1109
+ }
1110
+ ]
1111
+ }
1112
+ }
1113
+ }
1114
+ },
1115
+ "aggregates": {
1116
+ "type": "object",
1117
+ "additionalProperties": false,
1118
+ "required": [
1119
+ "with_skill",
1120
+ "baseline",
1121
+ "band_rule"
1122
+ ],
1123
+ "properties": {
1124
+ "with_skill": {
1125
+ "$ref": "#/$defs/modeAggregate"
1126
+ },
1127
+ "baseline": {
1128
+ "$ref": "#/$defs/modeAggregate"
1129
+ },
1130
+ "excluded_cases": {
1131
+ "type": "array",
1132
+ "description": "Cases excluded from BOTH arms of the aggregate because at least one of their arms had no measured result (spec 017 AC-6). `comparison.delta` is paired by construction, so a case that cannot be measured on one side is removed from both rather than from one — otherwise the delta is a mean over one case set minus a mean over another. The case itself remains in results.cases: it is removed from the mean, not from the record. Absent when nothing was excluded.",
1133
+ "items": {
1134
+ "type": "object",
1135
+ "additionalProperties": true,
1136
+ "required": [
1137
+ "id",
1138
+ "reason"
1139
+ ],
1140
+ "properties": {
1141
+ "id": {
1142
+ "type": "string",
1143
+ "description": "The case id excluded from both arms."
1144
+ },
1145
+ "modes": {
1146
+ "type": "array",
1147
+ "items": {
1148
+ "enum": [
1149
+ "with_skill",
1150
+ "baseline"
1151
+ ]
1152
+ },
1153
+ "description": "Which arm or arms were unusable."
1154
+ },
1155
+ "reason": {
1156
+ "type": "string",
1157
+ "description": "What was observed. Names no cause the receipt cannot establish."
1158
+ }
1159
+ }
1160
+ }
1161
+ },
1162
+ "band_rule": {
1163
+ "type": "string",
1164
+ "minLength": 1,
1165
+ "description": "v0.6. The formula the aggregate band (each arm's stddev, and delta_uncertainty) is derived by, stated as a constant so a reader can recompute it from results.cases."
1166
+ }
1167
+ }
1168
+ }
1169
+ }
1170
+ },
1171
+ "comparison": {
1172
+ "type": "object",
1173
+ "additionalProperties": false,
1174
+ "required": [
1175
+ "with_skill_score",
1176
+ "baseline_score",
1177
+ "delta",
1178
+ "delta_uncertainty"
1179
+ ],
1180
+ "properties": {
1181
+ "with_skill_score": {
1182
+ "type": [
1183
+ "number",
1184
+ "null"
1185
+ ],
1186
+ "minimum": 0,
1187
+ "maximum": 1,
1188
+ "description": "Null ONLY beside delta_uncertainty_unavailable no_cases."
1189
+ },
1190
+ "baseline_score": {
1191
+ "type": [
1192
+ "number",
1193
+ "null"
1194
+ ],
1195
+ "minimum": 0,
1196
+ "maximum": 1,
1197
+ "description": "Null ONLY on an imported (DECLARED) receipt from a tool with no baseline mode — never a fabricated 0."
1198
+ },
1199
+ "delta": {
1200
+ "type": [
1201
+ "number",
1202
+ "null"
1203
+ ],
1204
+ "minimum": -1,
1205
+ "maximum": 1,
1206
+ "description": "Null when baseline_score is null (no baseline mode was run)."
1207
+ },
1208
+ "delta_uncertainty": {
1209
+ "type": [
1210
+ "number",
1211
+ "null"
1212
+ ],
1213
+ "minimum": 0,
1214
+ "description": "Combined uncertainty of the delta (quadrature sum of the two aggregate bands). Null when delta is null, or (v0.6) when either band is null, in which case delta_uncertainty_unavailable says why."
1215
+ },
1216
+ "delta_uncertainty_unavailable": {
1217
+ "type": "string",
1218
+ "enum": [
1219
+ "single_case",
1220
+ "no_cases"
1221
+ ],
1222
+ "description": "v0.6. Present exactly when delta_uncertainty is null on a run that measured: single_case = an arm has one included case, so its band cannot form; no_cases = no case was included on either arm."
1223
+ }
1224
+ },
1225
+ "allOf": [
1226
+ {
1227
+ "description": "v0.6: the reason and the null come together.",
1228
+ "if": {
1229
+ "required": [
1230
+ "delta_uncertainty_unavailable"
1231
+ ]
1232
+ },
1233
+ "then": {
1234
+ "properties": {
1235
+ "delta_uncertainty": {
1236
+ "type": "null"
1237
+ }
1238
+ }
1239
+ }
1240
+ }
1241
+ ]
1242
+ },
1243
+ "verification_level": {
1244
+ "type": "string",
1245
+ "description": "Community verification lattice. FORMAL is reserved/unimplemented in v0.3.",
1246
+ "enum": [
1247
+ "UNVERIFIED",
1248
+ "DECLARED",
1249
+ "TESTED"
1250
+ ]
1251
+ },
1252
+ "editorial_reviews": {
1253
+ "type": "array",
1254
+ "description": "Optional pointers to external one-shot editorial reviews of this skill (context only; not verification evidence).",
1255
+ "items": {
1256
+ "type": "object",
1257
+ "additionalProperties": false,
1258
+ "required": [
1259
+ "url",
1260
+ "source",
1261
+ "date"
1262
+ ],
1263
+ "properties": {
1264
+ "url": {
1265
+ "type": "string",
1266
+ "minLength": 1
1267
+ },
1268
+ "source": {
1269
+ "type": "string",
1270
+ "minLength": 1
1271
+ },
1272
+ "date": {
1273
+ "type": "string",
1274
+ "pattern": "^\\d{4}-\\d{2}-\\d{2}$"
1275
+ }
1276
+ }
1277
+ }
1278
+ },
1279
+ "receipt_hash": {
1280
+ "type": "string",
1281
+ "description": "sha256 (hex) of the canonical receipt JSON with this field omitted.",
1282
+ "pattern": "^[a-f0-9]{64}$"
1283
+ },
1284
+ "economics": {
1285
+ "type": "object",
1286
+ "additionalProperties": false,
1287
+ "description": "v0.4 derived economics. Computed from the per-case `usage` at `run.pricing_snapshot` prices. The three value axes (accuracy lift, cost, latency) live separately here and in the reports: there is deliberately NO composite value score, because collapsing axes with different units and different error bars would produce a number no reader could trace to evidence.",
1288
+ "required": [
1289
+ "basis",
1290
+ "with_skill",
1291
+ "baseline",
1292
+ "judge_excluded"
1293
+ ],
1294
+ "properties": {
1295
+ "basis": {
1296
+ "enum": [
1297
+ "metered",
1298
+ "metered-equivalent"
1299
+ ],
1300
+ "description": "metered = real spend on an api surface; metered-equivalent = what the same tokens would have cost on the metered API (subscription CLI surfaces, where actual spend is $0)."
1301
+ },
1302
+ "surface": {
1303
+ "type": "string"
1304
+ },
1305
+ "with_skill": {
1306
+ "type": "object",
1307
+ "additionalProperties": false,
1308
+ "properties": {
1309
+ "call_count": {
1310
+ "type": "integer",
1311
+ "minimum": 0
1312
+ },
1313
+ "mean_input_tokens": {
1314
+ "type": [
1315
+ "number",
1316
+ "null"
1317
+ ]
1318
+ },
1319
+ "mean_output_tokens": {
1320
+ "type": [
1321
+ "number",
1322
+ "null"
1323
+ ]
1324
+ },
1325
+ "mean_cost_usd_per_call": {
1326
+ "type": [
1327
+ "number",
1328
+ "null"
1329
+ ]
1330
+ },
1331
+ "median_wall_ms": {
1332
+ "type": [
1333
+ "number",
1334
+ "null"
1335
+ ]
1336
+ },
1337
+ "wall_ms_p25": {
1338
+ "type": [
1339
+ "number",
1340
+ "null"
1341
+ ]
1342
+ },
1343
+ "wall_ms_p75": {
1344
+ "type": [
1345
+ "number",
1346
+ "null"
1347
+ ]
1348
+ },
1349
+ "wall_ms_iqr": {
1350
+ "type": [
1351
+ "number",
1352
+ "null"
1353
+ ]
1354
+ }
1355
+ }
1356
+ },
1357
+ "baseline": {
1358
+ "type": "object",
1359
+ "additionalProperties": false,
1360
+ "properties": {
1361
+ "call_count": {
1362
+ "type": "integer",
1363
+ "minimum": 0
1364
+ },
1365
+ "mean_input_tokens": {
1366
+ "type": [
1367
+ "number",
1368
+ "null"
1369
+ ]
1370
+ },
1371
+ "mean_output_tokens": {
1372
+ "type": [
1373
+ "number",
1374
+ "null"
1375
+ ]
1376
+ },
1377
+ "mean_cost_usd_per_call": {
1378
+ "type": [
1379
+ "number",
1380
+ "null"
1381
+ ]
1382
+ },
1383
+ "median_wall_ms": {
1384
+ "type": [
1385
+ "number",
1386
+ "null"
1387
+ ]
1388
+ },
1389
+ "wall_ms_p25": {
1390
+ "type": [
1391
+ "number",
1392
+ "null"
1393
+ ]
1394
+ },
1395
+ "wall_ms_p75": {
1396
+ "type": [
1397
+ "number",
1398
+ "null"
1399
+ ]
1400
+ },
1401
+ "wall_ms_iqr": {
1402
+ "type": [
1403
+ "number",
1404
+ "null"
1405
+ ]
1406
+ }
1407
+ }
1408
+ },
1409
+ "skill_incremental_cost_usd_per_call": {
1410
+ "type": [
1411
+ "number",
1412
+ "null"
1413
+ ]
1414
+ },
1415
+ "skill_incremental_cost_usd_per_1k_calls": {
1416
+ "type": [
1417
+ "number",
1418
+ "null"
1419
+ ]
1420
+ },
1421
+ "output_tokens_delta": {
1422
+ "type": [
1423
+ "number",
1424
+ "null"
1425
+ ]
1426
+ },
1427
+ "median_wall_ms_delta": {
1428
+ "type": [
1429
+ "number",
1430
+ "null"
1431
+ ]
1432
+ },
1433
+ "judge_excluded": {
1434
+ "const": true,
1435
+ "description": "Structural guarantee: judge usage never enters any figure in this block. A receipt cannot claim otherwise."
1436
+ },
1437
+ "judge_overhead": {
1438
+ "type": "object",
1439
+ "additionalProperties": false,
1440
+ "properties": {
1441
+ "note": {
1442
+ "type": "string"
1443
+ },
1444
+ "total_cost_usd": {
1445
+ "type": [
1446
+ "number",
1447
+ "null"
1448
+ ]
1449
+ },
1450
+ "case_rows_measured": {
1451
+ "type": "integer",
1452
+ "minimum": 0
1453
+ }
1454
+ }
1455
+ },
1456
+ "notes": {
1457
+ "type": "object",
1458
+ "additionalProperties": false,
1459
+ "properties": {
1460
+ "absolute_cost": {
1461
+ "type": "string"
1462
+ },
1463
+ "cache_pricing": {
1464
+ "type": "string"
1465
+ },
1466
+ "latency": {
1467
+ "type": "string"
1468
+ }
1469
+ }
1470
+ }
1471
+ }
1472
+ },
1473
+ "generation_sampled": {
1474
+ "type": "boolean",
1475
+ "description": "CAPABILITY FLAG (v0.5). A receipt that carries across-draw statistics — any results.cases[] entry with a `generation` block — MUST declare `generation_sampled: true`, and a receipt that declares it MUST carry at least one AND MUST carry `suite.canary` (F-015-A: F-014-F named both blocks). ABSENT MEANS LEGACY: a v0.5 receipt that ran no generation sampling (an imported DECLARED receipt, for instance) omits this field and stays valid, and every v0.4-and-earlier receipt is unaffected. The flag exists because v0.5 otherwise let a receipt claim conformance while carrying none of what v0.5 adds (F-014-F); binding the requirement to what the receipt DECLARES rather than to its verification_level closes that without invalidating a single archived receipt."
1476
+ }
1477
+ },
1478
+ "$defs": {
1479
+ "modeAggregate": {
1480
+ "type": "object",
1481
+ "additionalProperties": false,
1482
+ "required": [
1483
+ "case_count",
1484
+ "pass_count",
1485
+ "mean_score",
1486
+ "stddev"
1487
+ ],
1488
+ "properties": {
1489
+ "case_count": {
1490
+ "type": "integer",
1491
+ "minimum": 0
1492
+ },
1493
+ "pass_count": {
1494
+ "type": "integer",
1495
+ "minimum": 0
1496
+ },
1497
+ "borderline_count": {
1498
+ "type": "integer",
1499
+ "minimum": 0
1500
+ },
1501
+ "mean_score": {
1502
+ "type": [
1503
+ "number",
1504
+ "null"
1505
+ ],
1506
+ "minimum": 0,
1507
+ "maximum": 1,
1508
+ "description": "Null ONLY when case_count is 0 (no included case): the mean of nothing is not a number."
1509
+ },
1510
+ "stddev": {
1511
+ "type": [
1512
+ "number",
1513
+ "null"
1514
+ ],
1515
+ "minimum": 0,
1516
+ "description": "Suite dispersion: the sample standard deviation (n-1) of the per-case means across the included cases of this arm. A reported summary stat; the drift headline is driven by per-case band-overlap verdicts, not this band. v0.6: null ONLY when case_count is below 2, where the formula cannot form; never 0 in that case."
1517
+ }
1518
+ },
1519
+ "allOf": [
1520
+ {
1521
+ "description": "v0.6: a band the formula can form cannot hide behind a null.",
1522
+ "if": {
1523
+ "required": [
1524
+ "case_count"
1525
+ ],
1526
+ "properties": {
1527
+ "case_count": {
1528
+ "minimum": 2
1529
+ }
1530
+ }
1531
+ },
1532
+ "then": {
1533
+ "properties": {
1534
+ "stddev": {
1535
+ "type": "number"
1536
+ }
1537
+ }
1538
+ }
1539
+ },
1540
+ {
1541
+ "description": "v0.6: a mean over at least one case is a number.",
1542
+ "if": {
1543
+ "required": [
1544
+ "case_count"
1545
+ ],
1546
+ "properties": {
1547
+ "case_count": {
1548
+ "minimum": 1
1549
+ }
1550
+ }
1551
+ },
1552
+ "then": {
1553
+ "properties": {
1554
+ "mean_score": {
1555
+ "type": "number"
1556
+ }
1557
+ }
1558
+ }
1559
+ }
1560
+ ]
1561
+ }
1562
+ }
1563
+ }