driftproof 0.5.0 → 0.7.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,7 +2,7 @@
2
2
  "$schema": "https://json-schema.org/draft/2020-12/schema",
3
3
  "$id": "https://driftproofhq.com/spec/receipt.schema.json",
4
4
  "title": "driftproof receipt",
5
- "description": "A signed, dated record of running one agent skill's eval suite with and without the skill on one model version, with sampled judge scores and confidence bands. Receipt spec v0.4 — additive over v0.3.1: adds per-case, per-arm generation `usage` (input/output/cached tokens + measured wall_ms) captured from the surfaces that report it, a separate per-case `judge_usage` (measurement overhead, EXCLUDED from every skill-value figure by construction — `economics.judge_excluded` is const true), a run-level `run.pricing_snapshot` freezing the registry prices the derived dollar figures were computed from (so a receipt keeps its meaning when prices later change), and a derived `economics` block (per-arm mean cost/call, skill incremental cost per call and per 1k calls, output-length delta, median wall_ms with IQR). The three value axes — accuracy lift, cost, latency — are recorded separately and NEVER combined into a composite score. All v0.3.1 semantics are unchanged and every prior receipt still validates against its own frozen schema (v0.1, v0.2, v0.3, v0.3.1). The TESTED tightening (see allOf) is unchanged: the interop relaxations remain available only below TESTED.",
5
+ "description": "A hash-verified, dated record of running one agent skill eval suite with and without the skill on one model version, with the GENERATION sampled n times per arm and the judge sampled k times inside each draw. Receipt spec v0.5 — additive over v0.4: adds per-case `generation` carrying the full draw list (each draw with its own generation_hash, nested judge samples, mean and stddev), the across-draw mean and sd, the mean judge-level sd, and the variance_ratio between them (null when the judge sd is zero, never a division result); adds the sampling policy actually applied (n_planned, n_drawn, n_measured, n_unmeasured, stopping_reason); records a timed-out draw as status `unmeasured` with NO score, excluded from every statistic rather than counted as zero; and adds a per-suite canary. All v0.4 semantics are unchanged and every prior receipt still validates against its own frozen schema (v0.1, v0.2, v0.3, v0.3.1, v0.4).",
6
6
  "type": "object",
7
7
  "additionalProperties": false,
8
8
  "required": [
@@ -115,12 +115,108 @@
115
115
  }
116
116
  }
117
117
  }
118
+ },
119
+ {
120
+ "$comment": "driftproof/generation-sampled-declared",
121
+ "description": "F-014-F, first direction: a receipt carrying a draw set or a variance ratio must declare the capability. Without this a third-party emitter validates as v0.5 while carrying none of what v0.5 exists to add.",
122
+ "if": {
123
+ "required": [
124
+ "results"
125
+ ],
126
+ "properties": {
127
+ "results": {
128
+ "required": [
129
+ "cases"
130
+ ],
131
+ "properties": {
132
+ "cases": {
133
+ "contains": {
134
+ "required": [
135
+ "generation"
136
+ ]
137
+ }
138
+ }
139
+ }
140
+ }
141
+ }
142
+ },
143
+ "then": {
144
+ "required": [
145
+ "generation_sampled"
146
+ ],
147
+ "properties": {
148
+ "generation_sampled": {
149
+ "const": true
150
+ }
151
+ }
152
+ }
153
+ },
154
+ {
155
+ "$comment": "driftproof/generation-sampled-honest",
156
+ "description": "F-014-F, second direction: a receipt that declares the capability must carry it. A flag assertable by a receipt carrying nothing would be a second way to claim a capability falsely, and would close the exposure in one direction only.",
157
+ "if": {
158
+ "required": [
159
+ "generation_sampled"
160
+ ],
161
+ "properties": {
162
+ "generation_sampled": {
163
+ "const": true
164
+ }
165
+ }
166
+ },
167
+ "then": {
168
+ "required": [
169
+ "results"
170
+ ],
171
+ "properties": {
172
+ "results": {
173
+ "required": [
174
+ "cases"
175
+ ],
176
+ "properties": {
177
+ "cases": {
178
+ "contains": {
179
+ "required": [
180
+ "generation"
181
+ ]
182
+ }
183
+ }
184
+ }
185
+ }
186
+ }
187
+ }
188
+ },
189
+ {
190
+ "$comment": "driftproof/generation-sampled-canary",
191
+ "description": "F-015-A, the other half of F-014-F: a receipt declaring the capability must also carry the per-suite canary. F-014-F named BOTH results.cases[].generation and suite.canary; spec 015 bound the first and left this one, so a receipt could still claim v0.5 conformance while carrying only half of what v0.5 adds. A receipt that declares generation_sampled has by construction run a suite of ours, so it has a canary to record; omitting it is the same false claim in the other half. Legacy is untouched: a v0.4-and-earlier receipt is governed by its own frozen schema, and a v0.5 receipt that ran no generation sampling declares nothing and is unaffected.",
192
+ "if": {
193
+ "required": [
194
+ "generation_sampled"
195
+ ],
196
+ "properties": {
197
+ "generation_sampled": {
198
+ "const": true
199
+ }
200
+ }
201
+ },
202
+ "then": {
203
+ "required": [
204
+ "suite"
205
+ ],
206
+ "properties": {
207
+ "suite": {
208
+ "required": [
209
+ "canary"
210
+ ]
211
+ }
212
+ }
213
+ }
118
214
  }
119
215
  ],
120
216
  "properties": {
121
217
  "schema_version": {
122
218
  "type": "string",
123
- "const": "0.4"
219
+ "const": "0.5"
124
220
  },
125
221
  "skill": {
126
222
  "type": "object",
@@ -179,6 +275,11 @@
179
275
  "case_count": {
180
276
  "type": "integer",
181
277
  "minimum": 0
278
+ },
279
+ "canary": {
280
+ "type": "string",
281
+ "pattern": "^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$",
282
+ "description": "Per-suite canary GUID: a leaked suite is detectable in a corpus. Detection aid, not a control."
182
283
  }
183
284
  }
184
285
  },
@@ -247,7 +348,7 @@
247
348
  "failed_case_count": {
248
349
  "type": "integer",
249
350
  "minimum": 0,
250
- "description": "v0.3.1 (optional). Number of cases marked failed_timeout (excluded from aggregates)."
351
+ "description": "v0.3.1 (optional; redefined in v0.5). The number of CASES excluded from the aggregates: exactly the length of results.aggregates.excluded_cases, computed from that list so the two cannot disagree. Exclusion is PAIRWISE since v0.5 — a case with any unmeasured arm leaves BOTH arms together and is counted ONCE here, however many of its arms failed."
251
352
  },
252
353
  "runner_version": {
253
354
  "type": "string",
@@ -626,6 +727,226 @@
626
727
  "minimum": 0
627
728
  }
628
729
  }
730
+ },
731
+ "generation": {
732
+ "type": "object",
733
+ "additionalProperties": true,
734
+ "required": [
735
+ "n_drawn",
736
+ "n_measured",
737
+ "n_unmeasured",
738
+ "stopping_reason",
739
+ "draws"
740
+ ],
741
+ "properties": {
742
+ "n_planned": {
743
+ "type": "integer",
744
+ "minimum": 1
745
+ },
746
+ "n_drawn": {
747
+ "type": "integer",
748
+ "minimum": 0
749
+ },
750
+ "n_measured": {
751
+ "type": "integer",
752
+ "minimum": 0
753
+ },
754
+ "n_unmeasured": {
755
+ "type": "integer",
756
+ "minimum": 0
757
+ },
758
+ "stopping_reason": {
759
+ "enum": [
760
+ "min_reached",
761
+ "stabilised",
762
+ "max_reached",
763
+ "unmeasured_exhausted",
764
+ "below_min",
765
+ "escalating"
766
+ ]
767
+ },
768
+ "mean": {
769
+ "type": [
770
+ "number",
771
+ "null"
772
+ ]
773
+ },
774
+ "sd": {
775
+ "type": [
776
+ "number",
777
+ "null"
778
+ ]
779
+ },
780
+ "judge_sd_mean": {
781
+ "type": [
782
+ "number",
783
+ "null"
784
+ ]
785
+ },
786
+ "variance_ratio": {
787
+ "type": [
788
+ "number",
789
+ "null"
790
+ ]
791
+ },
792
+ "draws": {
793
+ "type": "array",
794
+ "items": {
795
+ "type": "object",
796
+ "additionalProperties": true,
797
+ "required": [
798
+ "draw_index",
799
+ "status"
800
+ ],
801
+ "properties": {
802
+ "draw_index": {
803
+ "type": "integer",
804
+ "minimum": 0
805
+ },
806
+ "status": {
807
+ "enum": [
808
+ "measured",
809
+ "unmeasured"
810
+ ]
811
+ },
812
+ "generation_hash": {
813
+ "type": [
814
+ "string",
815
+ "null"
816
+ ]
817
+ },
818
+ "samples": {
819
+ "type": "array",
820
+ "items": {
821
+ "type": "number"
822
+ }
823
+ },
824
+ "judge_sample_hashes": {
825
+ "type": "array",
826
+ "items": {
827
+ "type": "string"
828
+ }
829
+ },
830
+ "mean": {
831
+ "type": [
832
+ "number",
833
+ "null"
834
+ ]
835
+ },
836
+ "stddev": {
837
+ "type": [
838
+ "number",
839
+ "null"
840
+ ]
841
+ },
842
+ "reason": {
843
+ "type": "string"
844
+ },
845
+ "usage": {
846
+ "type": "object"
847
+ },
848
+ "judge_usage": {
849
+ "type": "object"
850
+ }
851
+ },
852
+ "allOf": [
853
+ {
854
+ "if": {
855
+ "properties": {
856
+ "status": {
857
+ "const": "unmeasured"
858
+ }
859
+ },
860
+ "required": [
861
+ "status"
862
+ ]
863
+ },
864
+ "then": {
865
+ "properties": {
866
+ "mean": {
867
+ "type": "null"
868
+ },
869
+ "stddev": {
870
+ "type": "null"
871
+ },
872
+ "samples": {
873
+ "maxItems": 0
874
+ }
875
+ }
876
+ }
877
+ },
878
+ {
879
+ "if": {
880
+ "properties": {
881
+ "status": {
882
+ "const": "measured"
883
+ }
884
+ },
885
+ "required": [
886
+ "status"
887
+ ]
888
+ },
889
+ "then": {
890
+ "required": [
891
+ "generation_hash",
892
+ "samples",
893
+ "mean",
894
+ "stddev"
895
+ ],
896
+ "properties": {
897
+ "mean": {
898
+ "type": "number"
899
+ },
900
+ "generation_hash": {
901
+ "type": "string"
902
+ }
903
+ }
904
+ }
905
+ }
906
+ ]
907
+ }
908
+ },
909
+ "variance_ratio_unavailable": {
910
+ "type": [
911
+ "string",
912
+ "null"
913
+ ],
914
+ "enum": [
915
+ "single_judge_sample",
916
+ "judge_sd_zero",
917
+ "no_measured_draws",
918
+ "judge_samples_unknown",
919
+ null
920
+ ],
921
+ "description": "WHICH null the variance_ratio is (F-014-C). `null` when a ratio was formed. `single_judge_sample`: k=1, so the per-draw judge spread is 0 by construction and the ratio is undefined — this is the shape every pre-v0.5 example produced. `judge_sd_zero`: k>=2 and the judge agreed with itself perfectly inside every measured draw. `no_measured_draws`: nothing was measured. `judge_samples_unknown`: the draws carry no sample list, so the count cannot be established and saying which null it is would assert a cause this control cannot reach. Each names what was OBSERVED and none names a cause."
922
+ }
923
+ },
924
+ "allOf": [
925
+ {
926
+ "$comment": "driftproof/variance-ratio-cause",
927
+ "description": "F-014-C: a null ratio must say WHICH null it is. Bound to the null rather than required outright, so a receipt that formed a ratio carries no reason and nothing in the archive is retroactively incomplete.",
928
+ "if": {
929
+ "required": [
930
+ "variance_ratio"
931
+ ],
932
+ "properties": {
933
+ "variance_ratio": {
934
+ "type": "null"
935
+ }
936
+ }
937
+ },
938
+ "then": {
939
+ "required": [
940
+ "variance_ratio_unavailable"
941
+ ],
942
+ "properties": {
943
+ "variance_ratio_unavailable": {
944
+ "type": "string"
945
+ }
946
+ }
947
+ }
948
+ }
949
+ ]
629
950
  }
630
951
  }
631
952
  }
@@ -643,6 +964,38 @@
643
964
  },
644
965
  "baseline": {
645
966
  "$ref": "#/$defs/modeAggregate"
967
+ },
968
+ "excluded_cases": {
969
+ "type": "array",
970
+ "description": "Cases excluded from BOTH arms of the aggregate because at least one of their arms had no measured result (spec 017 AC-6). `comparison.delta` is paired by construction, so a case that cannot be measured on one side is removed from both rather than from one — otherwise the delta is a mean over one case set minus a mean over another. The case itself remains in results.cases: it is removed from the mean, not from the record. Absent when nothing was excluded.",
971
+ "items": {
972
+ "type": "object",
973
+ "additionalProperties": true,
974
+ "required": [
975
+ "id",
976
+ "reason"
977
+ ],
978
+ "properties": {
979
+ "id": {
980
+ "type": "string",
981
+ "description": "The case id excluded from both arms."
982
+ },
983
+ "modes": {
984
+ "type": "array",
985
+ "items": {
986
+ "enum": [
987
+ "with_skill",
988
+ "baseline"
989
+ ]
990
+ },
991
+ "description": "Which arm or arms were unusable."
992
+ },
993
+ "reason": {
994
+ "type": "string",
995
+ "description": "What was observed. Names no cause the receipt cannot establish."
996
+ }
997
+ }
998
+ }
646
999
  }
647
1000
  }
648
1001
  }
@@ -920,6 +1273,10 @@
920
1273
  }
921
1274
  }
922
1275
  }
1276
+ },
1277
+ "generation_sampled": {
1278
+ "type": "boolean",
1279
+ "description": "CAPABILITY FLAG (v0.5). A receipt that carries across-draw statistics — any results.cases[] entry with a `generation` block — MUST declare `generation_sampled: true`, and a receipt that declares it MUST carry at least one AND MUST carry `suite.canary` (F-015-A: F-014-F named both blocks). ABSENT MEANS LEGACY: a v0.5 receipt that ran no generation sampling (an imported DECLARED receipt, for instance) omits this field and stays valid, and every v0.4-and-earlier receipt is unaffected. The flag exists because v0.5 otherwise let a receipt claim conformance while carrying none of what v0.5 adds (F-014-F); binding the requirement to what the receipt DECLARES rather than to its verification_level closes that without invalidating a single archived receipt."
923
1280
  }
924
1281
  },
925
1282
  "$defs": {