driftproof 0.8.1 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,7 +2,7 @@
2
2
  "$schema": "https://json-schema.org/draft/2020-12/schema",
3
3
  "$id": "https://driftproofhq.com/spec/receipt.schema.json",
4
4
  "title": "driftproof receipt",
5
- "description": "A hash-verified, dated record of running one agent skill eval suite with and without the skill on one model version, with the GENERATION sampled n times per arm and the judge sampled k times inside each draw. Receipt spec v0.5 — additive over v0.4: adds per-case `generation` carrying the full draw list (each draw with its own generation_hash, nested judge samples, mean and stddev), the across-draw mean and sd, the mean judge-level sd, and the variance_ratio between them (null when the judge sd is zero, never a division result); adds the sampling policy actually applied (n_planned, n_drawn, n_measured, n_unmeasured, stopping_reason); records a timed-out draw as status `unmeasured` with NO score, excluded from every statistic rather than counted as zero; and adds a per-suite canary. All v0.4 semantics are unchanged and every prior receipt still validates against its own frozen schema (v0.1, v0.2, v0.3, v0.3.1, v0.4).",
5
+ "description": "A hash-verified, dated record of running one agent skill eval suite with and without the skill on one model version, with the GENERATION sampled n times per arm and the judge sampled k times inside each draw. Receipt spec v0.6 — over v0.5, the receipt says what answered it: `run.answered_by` (kind model|stub|external, whether the surface echoed a model id and which, and which spawn path produced it); `run.surface` and `run.judge.surface` may read `stub` (nothing answered; the text was canned) and a TESTED receipt requires answered_by.kind model; `run.judge.model_id` and `run.judge.prompt_template_hash` say which judge ran; every draw records its `stop_reason` and whether it was `truncated` (a truncated draw is unmeasured, never judged) and the case counts `n_truncated`; `case_status` gains `failed_unmeasured` (every draw of the arm was unmeasured for a reason that is not a timeout), recorded without fabricated samples or hashes exactly as a timeout is; `results.aggregates.band_rule` states the formula the aggregate band is derived by; an aggregate `stddev` is null only when its arm has fewer than two cases and `comparison.delta_uncertainty` is null only beside `delta_uncertainty_unavailable`. Every prior receipt still validates against its own frozen schema (v0.1, v0.2, v0.3, v0.3.1, v0.4, v0.5).",
6
6
  "type": "object",
7
7
  "additionalProperties": false,
8
8
  "required": [
@@ -17,7 +17,7 @@
17
17
  ],
18
18
  "allOf": [
19
19
  {
20
- "description": "TESTED tightening — the interop relaxations (null hashes, external surface, null comparison, transcripts 'none') are ONLY available to receipts below TESTED. A TESTED receipt must carry the full evidence chain, exactly as before the interop revision.",
20
+ "description": "TESTED tightening — the interop relaxations (null hashes, external surface, null comparison, transcripts 'none') are ONLY available to receipts below TESTED. A TESTED receipt must carry the full evidence chain, exactly as before the interop revision. v0.6: a TESTED receipt was answered by a model (answered_by.kind model), its judge template is recorded, and its comparison is numeric except where delta_uncertainty_unavailable names why a band could not form.",
21
21
  "if": {
22
22
  "required": [
23
23
  "verification_level"
@@ -62,37 +62,95 @@
62
62
  "retained-local",
63
63
  "hashes-only"
64
64
  ]
65
+ },
66
+ "answered_by": {
67
+ "properties": {
68
+ "kind": {
69
+ "const": "model"
70
+ }
71
+ }
72
+ },
73
+ "judge": {
74
+ "properties": {
75
+ "prompt_template_hash": {
76
+ "type": "string"
77
+ }
78
+ }
65
79
  }
66
80
  }
67
81
  },
68
82
  "comparison": {
69
- "properties": {
70
- "baseline_score": {
71
- "type": "number"
72
- },
73
- "delta": {
74
- "type": "number"
83
+ "allOf": [
84
+ {
85
+ "if": {
86
+ "not": {
87
+ "required": [
88
+ "delta_uncertainty_unavailable"
89
+ ]
90
+ }
91
+ },
92
+ "then": {
93
+ "properties": {
94
+ "baseline_score": {
95
+ "type": "number"
96
+ },
97
+ "delta": {
98
+ "type": "number"
99
+ },
100
+ "delta_uncertainty": {
101
+ "type": "number"
102
+ }
103
+ }
104
+ }
75
105
  },
76
- "delta_uncertainty": {
77
- "type": "number"
106
+ {
107
+ "if": {
108
+ "required": [
109
+ "delta_uncertainty_unavailable"
110
+ ],
111
+ "properties": {
112
+ "delta_uncertainty_unavailable": {
113
+ "const": "single_case"
114
+ }
115
+ }
116
+ },
117
+ "then": {
118
+ "properties": {
119
+ "with_skill_score": {
120
+ "type": "number"
121
+ },
122
+ "baseline_score": {
123
+ "type": "number"
124
+ },
125
+ "delta": {
126
+ "type": "number"
127
+ }
128
+ }
129
+ }
78
130
  }
79
- }
131
+ ]
80
132
  },
81
133
  "results": {
82
134
  "properties": {
83
135
  "cases": {
84
136
  "items": {
85
137
  "if": {
86
- "not": {
87
- "required": [
88
- "case_status"
89
- ],
90
- "properties": {
91
- "case_status": {
92
- "const": "failed_timeout"
138
+ "anyOf": [
139
+ {
140
+ "not": {
141
+ "required": [
142
+ "case_status"
143
+ ]
144
+ }
145
+ },
146
+ {
147
+ "properties": {
148
+ "case_status": {
149
+ "const": "ok"
150
+ }
93
151
  }
94
152
  }
95
- }
153
+ ]
96
154
  },
97
155
  "then": {
98
156
  "required": [
@@ -216,7 +274,7 @@
216
274
  "properties": {
217
275
  "schema_version": {
218
276
  "type": "string",
219
- "const": "0.5"
277
+ "const": "0.6"
220
278
  },
221
279
  "skill": {
222
280
  "type": "object",
@@ -294,7 +352,8 @@
294
352
  "date_utc",
295
353
  "judge",
296
354
  "registry",
297
- "transcripts"
355
+ "transcripts",
356
+ "answered_by"
298
357
  ],
299
358
  "properties": {
300
359
  "model_id": {
@@ -324,9 +383,10 @@
324
383
  "claude-cli",
325
384
  "openai-api",
326
385
  "openai-cli",
327
- "external"
386
+ "external",
387
+ "stub"
328
388
  ],
329
- "description": "'external' = the run happened on another tool's harness and was imported (never valid on a TESTED receipt)."
389
+ "description": "'external' = the run happened on another tool's harness and was imported (never valid on a TESTED receipt). 'stub' = nothing answered (DRIFTPROOF_STUB=1); the text was canned. Never valid on a TESTED receipt."
330
390
  },
331
391
  "source": {
332
392
  "type": "string",
@@ -383,7 +443,9 @@
383
443
  "required": [
384
444
  "samples",
385
445
  "temperature",
386
- "sampling"
446
+ "sampling",
447
+ "model_id",
448
+ "prompt_template_hash"
387
449
  ],
388
450
  "properties": {
389
451
  "samples": {
@@ -409,8 +471,22 @@
409
471
  "claude-cli",
410
472
  "openai-api",
411
473
  "openai-cli",
412
- "external"
474
+ "external",
475
+ "stub"
413
476
  ]
477
+ },
478
+ "model_id": {
479
+ "type": "string",
480
+ "minLength": 1,
481
+ "description": "v0.6. The judge model that graded every case of this run; every case's `judge.model_id` equals it on a receipt whose answered_by.kind is model."
482
+ },
483
+ "prompt_template_hash": {
484
+ "type": [
485
+ "string",
486
+ "null"
487
+ ],
488
+ "pattern": "^[a-f0-9]{64}$",
489
+ "description": "v0.6. sha256 over the grading template with its three slots (rubric, task, response) empty: a different template is a different judge. Null ONLY on an imported receipt (answered_by.kind external)."
414
490
  }
415
491
  }
416
492
  },
@@ -461,6 +537,56 @@
461
537
  }
462
538
  }
463
539
  }
540
+ },
541
+ "answered_by": {
542
+ "type": "object",
543
+ "additionalProperties": false,
544
+ "description": "v0.6. What answered the run. `kind` model = a model surface answered; stub = the canned stub answered, nothing was measured; external = another tool's harness answered and the receipt was imported. `attested` = the surface echoed a model id for every draw and it matched the requested canonical id. `reported_model` = the id the surface echoed (null when it echoed nothing); `reported_models` = every id the surface named across the run (null when none). `isolation` = the spawn path: eval-user (the isolated hop), same-user (--trusted-skill), none (the stub, or an api surface that spawns nothing).",
545
+ "required": [
546
+ "kind",
547
+ "attested",
548
+ "reported_model",
549
+ "reported_models",
550
+ "isolation"
551
+ ],
552
+ "properties": {
553
+ "kind": {
554
+ "type": "string",
555
+ "enum": [
556
+ "model",
557
+ "stub",
558
+ "external"
559
+ ]
560
+ },
561
+ "attested": {
562
+ "type": "boolean"
563
+ },
564
+ "reported_model": {
565
+ "type": [
566
+ "string",
567
+ "null"
568
+ ],
569
+ "minLength": 1
570
+ },
571
+ "reported_models": {
572
+ "type": [
573
+ "array",
574
+ "null"
575
+ ],
576
+ "items": {
577
+ "type": "string",
578
+ "minLength": 1
579
+ }
580
+ },
581
+ "isolation": {
582
+ "type": "string",
583
+ "enum": [
584
+ "eval-user",
585
+ "same-user",
586
+ "none"
587
+ ]
588
+ }
589
+ }
464
590
  }
465
591
  }
466
592
  },
@@ -483,25 +609,26 @@
483
609
  ],
484
610
  "allOf": [
485
611
  {
486
- "description": "A completed case carries the full sampled band + hashes; a failed_timeout case is recorded WITHOUT fabricated samples (it is excluded from aggregates).",
612
+ "description": "A completed case carries the full sampled band + hashes; a failed case (any case_status that is not ok) is recorded WITHOUT fabricated samples (it is excluded from aggregates). v0.6 keys this on case_status not ok rather than on the failed_timeout literal.",
487
613
  "if": {
488
- "required": [
489
- "case_status"
490
- ],
491
- "properties": {
492
- "case_status": {
493
- "const": "failed_timeout"
614
+ "anyOf": [
615
+ {
616
+ "not": {
617
+ "required": [
618
+ "case_status"
619
+ ]
620
+ }
621
+ },
622
+ {
623
+ "properties": {
624
+ "case_status": {
625
+ "const": "ok"
626
+ }
627
+ }
494
628
  }
495
- }
496
- },
497
- "then": {
498
- "required": [
499
- "id",
500
- "mode",
501
- "case_status"
502
629
  ]
503
630
  },
504
- "else": {
631
+ "then": {
505
632
  "required": [
506
633
  "outcome",
507
634
  "score",
@@ -510,6 +637,13 @@
510
637
  "samples",
511
638
  "judge"
512
639
  ]
640
+ },
641
+ "else": {
642
+ "required": [
643
+ "id",
644
+ "mode",
645
+ "case_status"
646
+ ]
513
647
  }
514
648
  }
515
649
  ],
@@ -527,10 +661,11 @@
527
661
  },
528
662
  "case_status": {
529
663
  "type": "string",
530
- "description": "v0.3.1 (optional; default 'ok'). 'failed_timeout' = the case's model/judge call persistently timed out after retries; the case is recorded but EXCLUDED from aggregates/verdicts — no samples/hashes are fabricated for it.",
664
+ "description": "v0.3.1 (optional; default 'ok'). 'failed_timeout' = the case's model/judge call persistently timed out after retries. v0.6: 'failed_unmeasured' = every draw of the arm was unmeasured for a reason that is not a timeout (an empty generation, a judge output with no score, a truncated draw). Either way the case is recorded but EXCLUDED from aggregates/verdicts — no samples/hashes are fabricated for it.",
531
665
  "enum": [
532
666
  "ok",
533
- "failed_timeout"
667
+ "failed_timeout",
668
+ "failed_unmeasured"
534
669
  ]
535
670
  },
536
671
  "outcome": {
@@ -736,7 +871,8 @@
736
871
  "n_measured",
737
872
  "n_unmeasured",
738
873
  "stopping_reason",
739
- "draws"
874
+ "draws",
875
+ "n_truncated"
740
876
  ],
741
877
  "properties": {
742
878
  "n_planned": {
@@ -796,7 +932,9 @@
796
932
  "additionalProperties": true,
797
933
  "required": [
798
934
  "draw_index",
799
- "status"
935
+ "status",
936
+ "stop_reason",
937
+ "truncated"
800
938
  ],
801
939
  "properties": {
802
940
  "draw_index": {
@@ -847,6 +985,24 @@
847
985
  },
848
986
  "judge_usage": {
849
987
  "type": "object"
988
+ },
989
+ "stop_reason": {
990
+ "type": [
991
+ "string",
992
+ "null"
993
+ ],
994
+ "description": "v0.6. Why the generation stopped, as the surface reported it (e.g. end_turn, max_tokens); null when the surface reports none."
995
+ },
996
+ "truncated": {
997
+ "type": "boolean",
998
+ "description": "v0.6. True when the generation was cut at the output cap. A truncated draw is unmeasured."
999
+ },
1000
+ "reported_model": {
1001
+ "type": [
1002
+ "string",
1003
+ "null"
1004
+ ],
1005
+ "description": "v0.6. The model id the surface echoed for this draw; null when it echoed nothing."
850
1006
  }
851
1007
  },
852
1008
  "allOf": [
@@ -919,6 +1075,11 @@
919
1075
  null
920
1076
  ],
921
1077
  "description": "WHICH null the variance_ratio is (F-014-C). `null` when a ratio was formed. `single_judge_sample`: k=1, so the per-draw judge spread is 0 by construction and the ratio is undefined — this is the shape every pre-v0.5 example produced. `judge_sd_zero`: k>=2 and the judge agreed with itself perfectly inside every measured draw. `no_measured_draws`: nothing was measured. `judge_samples_unknown`: the draws carry no sample list, so the count cannot be established and saying which null it is would assert a cause this control cannot reach. Each names what was OBSERVED and none names a cause."
1078
+ },
1079
+ "n_truncated": {
1080
+ "type": "integer",
1081
+ "minimum": 0,
1082
+ "description": "v0.6. Draws whose generation stopped at the output cap (stop_reason max_tokens or the surface's equivalent); each is unmeasured, never judged, and excluded from every statistic."
922
1083
  }
923
1084
  },
924
1085
  "allOf": [
@@ -956,7 +1117,8 @@
956
1117
  "additionalProperties": false,
957
1118
  "required": [
958
1119
  "with_skill",
959
- "baseline"
1120
+ "baseline",
1121
+ "band_rule"
960
1122
  ],
961
1123
  "properties": {
962
1124
  "with_skill": {
@@ -996,6 +1158,11 @@
996
1158
  }
997
1159
  }
998
1160
  }
1161
+ },
1162
+ "band_rule": {
1163
+ "type": "string",
1164
+ "minLength": 1,
1165
+ "description": "v0.6. The formula the aggregate band (each arm's stddev, and delta_uncertainty) is derived by, stated as a constant so a reader can recompute it from results.cases."
999
1166
  }
1000
1167
  }
1001
1168
  }
@@ -1012,9 +1179,13 @@
1012
1179
  ],
1013
1180
  "properties": {
1014
1181
  "with_skill_score": {
1015
- "type": "number",
1182
+ "type": [
1183
+ "number",
1184
+ "null"
1185
+ ],
1016
1186
  "minimum": 0,
1017
- "maximum": 1
1187
+ "maximum": 1,
1188
+ "description": "Null ONLY beside delta_uncertainty_unavailable no_cases."
1018
1189
  },
1019
1190
  "baseline_score": {
1020
1191
  "type": [
@@ -1040,9 +1211,34 @@
1040
1211
  "null"
1041
1212
  ],
1042
1213
  "minimum": 0,
1043
- "description": "Combined uncertainty of the delta (quadrature sum of the two aggregate bands). Null when delta is null."
1214
+ "description": "Combined uncertainty of the delta (quadrature sum of the two aggregate bands). Null when delta is null, or (v0.6) when either band is null, in which case delta_uncertainty_unavailable says why."
1215
+ },
1216
+ "delta_uncertainty_unavailable": {
1217
+ "type": "string",
1218
+ "enum": [
1219
+ "single_case",
1220
+ "no_cases"
1221
+ ],
1222
+ "description": "v0.6. Present exactly when delta_uncertainty is null on a run that measured: single_case = an arm has one included case, so its band cannot form; no_cases = no case was included on either arm."
1044
1223
  }
1045
- }
1224
+ },
1225
+ "allOf": [
1226
+ {
1227
+ "description": "v0.6: the reason and the null come together.",
1228
+ "if": {
1229
+ "required": [
1230
+ "delta_uncertainty_unavailable"
1231
+ ]
1232
+ },
1233
+ "then": {
1234
+ "properties": {
1235
+ "delta_uncertainty": {
1236
+ "type": "null"
1237
+ }
1238
+ }
1239
+ }
1240
+ }
1241
+ ]
1046
1242
  },
1047
1243
  "verification_level": {
1048
1244
  "type": "string",
@@ -1303,16 +1499,65 @@
1303
1499
  "minimum": 0
1304
1500
  },
1305
1501
  "mean_score": {
1306
- "type": "number",
1502
+ "type": [
1503
+ "number",
1504
+ "null"
1505
+ ],
1307
1506
  "minimum": 0,
1308
- "maximum": 1
1507
+ "maximum": 1,
1508
+ "description": "Null ONLY when case_count is 0 (no included case): the mean of nothing is not a number."
1309
1509
  },
1310
1510
  "stddev": {
1311
- "type": "number",
1511
+ "type": [
1512
+ "number",
1513
+ "null"
1514
+ ],
1312
1515
  "minimum": 0,
1313
- "description": "Suite dispersion: stddev of the per-case means across the suite. A reported summary stat; the drift headline is driven by per-case band-overlap verdicts, not this band."
1516
+ "description": "Suite dispersion: the sample standard deviation (n-1) of the per-case means across the included cases of this arm. A reported summary stat; the drift headline is driven by per-case band-overlap verdicts, not this band. v0.6: null ONLY when case_count is below 2, where the formula cannot form; never 0 in that case."
1314
1517
  }
1315
- }
1518
+ },
1519
+ "allOf": [
1520
+ {
1521
+ "description": "v0.6: a band the formula can form cannot hide behind a null.",
1522
+ "if": {
1523
+ "required": [
1524
+ "case_count"
1525
+ ],
1526
+ "properties": {
1527
+ "case_count": {
1528
+ "minimum": 2
1529
+ }
1530
+ }
1531
+ },
1532
+ "then": {
1533
+ "properties": {
1534
+ "stddev": {
1535
+ "type": "number"
1536
+ }
1537
+ }
1538
+ }
1539
+ },
1540
+ {
1541
+ "description": "v0.6: a mean over at least one case is a number.",
1542
+ "if": {
1543
+ "required": [
1544
+ "case_count"
1545
+ ],
1546
+ "properties": {
1547
+ "case_count": {
1548
+ "minimum": 1
1549
+ }
1550
+ }
1551
+ },
1552
+ "then": {
1553
+ "properties": {
1554
+ "mean_score": {
1555
+ "type": "number"
1556
+ }
1557
+ }
1558
+ }
1559
+ }
1560
+ ]
1316
1561
  }
1317
1562
  }
1318
1563
  }