driftproof 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,642 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://driftproofhq.com/spec/receipt.v0.3.1.schema.json",
4
+ "title": "driftproof receipt",
5
+ "description": "FROZEN receipt spec v0.3.1. A signed, dated record of running one agent skill's eval suite with and without the skill on one model version, with sampled judge scores and confidence bands. Receipt spec v0.3.1 — additive over v0.3: adds run.provider (two-axis provider), expands the surface enum with the OpenAI lanes (openai-api, openai-cli), adds an optional run.surface_overhead_note (fixed harness preamble on the openai/cli surface), optional per-case deterministic post-check results, and an optional skill.tokens count for the value-per-token axis. Interop-additive revision: receipts IMPORTED from external tools (surface 'external', run.source 'imported/<tool>', verification_level DECLARED) may carry null content/suite/rubric hashes and omit generation hashes — hashes are never fabricated; the TESTED tightening (see allOf) requires the full evidence chain whenever verification_level is TESTED, so no previously issued receipt is invalidated and TESTED keeps its meaning.",
6
+ "type": "object",
7
+ "additionalProperties": false,
8
+ "required": [
9
+ "schema_version",
10
+ "skill",
11
+ "suite",
12
+ "run",
13
+ "results",
14
+ "comparison",
15
+ "verification_level",
16
+ "receipt_hash"
17
+ ],
18
+ "allOf": [
19
+ {
20
+ "description": "TESTED tightening — the interop relaxations (null hashes, external surface, null comparison, transcripts 'none') are ONLY available to receipts below TESTED. A TESTED receipt must carry the full evidence chain, exactly as before the interop revision.",
21
+ "if": {
22
+ "required": [
23
+ "verification_level"
24
+ ],
25
+ "properties": {
26
+ "verification_level": {
27
+ "const": "TESTED"
28
+ }
29
+ }
30
+ },
31
+ "then": {
32
+ "properties": {
33
+ "skill": {
34
+ "properties": {
35
+ "content_hash": {
36
+ "type": "string"
37
+ }
38
+ }
39
+ },
40
+ "suite": {
41
+ "properties": {
42
+ "format": {
43
+ "const": "agentskills.io/evals"
44
+ },
45
+ "suite_hash": {
46
+ "type": "string"
47
+ }
48
+ }
49
+ },
50
+ "run": {
51
+ "properties": {
52
+ "surface": {
53
+ "enum": [
54
+ "api",
55
+ "claude-cli",
56
+ "openai-api",
57
+ "openai-cli"
58
+ ]
59
+ },
60
+ "transcripts": {
61
+ "enum": [
62
+ "retained-local",
63
+ "hashes-only"
64
+ ]
65
+ }
66
+ }
67
+ },
68
+ "comparison": {
69
+ "properties": {
70
+ "baseline_score": {
71
+ "type": "number"
72
+ },
73
+ "delta": {
74
+ "type": "number"
75
+ },
76
+ "delta_uncertainty": {
77
+ "type": "number"
78
+ }
79
+ }
80
+ },
81
+ "results": {
82
+ "properties": {
83
+ "cases": {
84
+ "items": {
85
+ "if": {
86
+ "not": {
87
+ "required": [
88
+ "case_status"
89
+ ],
90
+ "properties": {
91
+ "case_status": {
92
+ "const": "failed_timeout"
93
+ }
94
+ }
95
+ }
96
+ },
97
+ "then": {
98
+ "required": [
99
+ "generation_hash",
100
+ "judge_sample_hashes"
101
+ ],
102
+ "properties": {
103
+ "judge": {
104
+ "properties": {
105
+ "rubric_hash": {
106
+ "type": "string"
107
+ }
108
+ }
109
+ }
110
+ }
111
+ }
112
+ }
113
+ }
114
+ }
115
+ }
116
+ }
117
+ }
118
+ }
119
+ ],
120
+ "properties": {
121
+ "schema_version": {
122
+ "type": "string",
123
+ "const": "0.3.1"
124
+ },
125
+ "skill": {
126
+ "type": "object",
127
+ "additionalProperties": false,
128
+ "required": [
129
+ "name",
130
+ "version",
131
+ "content_hash"
132
+ ],
133
+ "properties": {
134
+ "name": {
135
+ "type": "string",
136
+ "minLength": 1
137
+ },
138
+ "version": {
139
+ "type": "string",
140
+ "minLength": 1
141
+ },
142
+ "content_hash": {
143
+ "type": [
144
+ "string",
145
+ "null"
146
+ ],
147
+ "description": "sha256 (hex) over SKILL.md + all bundled files in canonical path-sorted order. Null ONLY on an imported (DECLARED) receipt whose skill bytes were never seen; a TESTED receipt must carry the hash (see the TESTED tightening).",
148
+ "pattern": "^[a-f0-9]{64}$"
149
+ },
150
+ "tokens": {
151
+ "type": "integer",
152
+ "minimum": 0,
153
+ "description": "v0.3.1 (optional). Estimated token size of the skill's SKILL.md (a coarse chars/4 proxy, not a model tokenizer), used for the value-per-token axis (delta per 1k skill tokens). See docs/methodology.html."
154
+ }
155
+ }
156
+ },
157
+ "suite": {
158
+ "type": "object",
159
+ "additionalProperties": false,
160
+ "required": [
161
+ "format",
162
+ "suite_hash",
163
+ "case_count"
164
+ ],
165
+ "properties": {
166
+ "format": {
167
+ "type": "string",
168
+ "minLength": 1,
169
+ "description": "Suite format. Driftproof-run (TESTED) receipts are always 'agentskills.io/evals' (see the TESTED tightening); an imported receipt names the source tool's format (e.g. 'skillgrade/eval.yaml')."
170
+ },
171
+ "suite_hash": {
172
+ "type": [
173
+ "string",
174
+ "null"
175
+ ],
176
+ "description": "sha256 (hex) over the canonicalized normalized case list. Null ONLY on an imported (DECLARED) receipt whose suite bytes were never seen.",
177
+ "pattern": "^[a-f0-9]{64}$"
178
+ },
179
+ "case_count": {
180
+ "type": "integer",
181
+ "minimum": 0
182
+ }
183
+ }
184
+ },
185
+ "run": {
186
+ "type": "object",
187
+ "additionalProperties": false,
188
+ "required": [
189
+ "model_id",
190
+ "provider",
191
+ "surface",
192
+ "runner_version",
193
+ "date_utc",
194
+ "judge",
195
+ "registry",
196
+ "transcripts"
197
+ ],
198
+ "properties": {
199
+ "model_id": {
200
+ "type": "string",
201
+ "minLength": 1
202
+ },
203
+ "model_release_date": {
204
+ "description": "ISO date (YYYY-MM-DD) of the model release if known, else null.",
205
+ "type": [
206
+ "string",
207
+ "null"
208
+ ],
209
+ "pattern": "^\\d{4}-\\d{2}-\\d{2}$"
210
+ },
211
+ "provider": {
212
+ "type": "string",
213
+ "description": "v0.3.1. The two-axis provider the target model ran on (registry `provider`, else inferred from the id).",
214
+ "enum": [
215
+ "anthropic",
216
+ "openai"
217
+ ]
218
+ },
219
+ "surface": {
220
+ "type": "string",
221
+ "enum": [
222
+ "api",
223
+ "claude-cli",
224
+ "openai-api",
225
+ "openai-cli",
226
+ "external"
227
+ ],
228
+ "description": "'external' = the run happened on another tool's harness and was imported (never valid on a TESTED receipt)."
229
+ },
230
+ "source": {
231
+ "type": "string",
232
+ "minLength": 1,
233
+ "description": "Interop-additive (optional). Provenance of a converted receipt, e.g. 'imported/agent-skills-eval' or 'imported/skillgrade'. Absent on receipts Driftproof ran itself."
234
+ },
235
+ "surface_overhead_note": {
236
+ "type": "string",
237
+ "description": "v0.3.1 (optional). Present on the openai/cli surface: states the fixed Codex base-instruction preamble (~12–15k input tokens per call) that the harness prepends and does not control."
238
+ },
239
+ "status": {
240
+ "type": "string",
241
+ "description": "v0.3.1 (optional; default 'complete'). 'incomplete' when >=1 case persistently failed (e.g. failed_timeout) and was EXCLUDED from aggregates. A drift/durability report must not compute a verdict from an incomplete receipt.",
242
+ "enum": [
243
+ "complete",
244
+ "incomplete"
245
+ ]
246
+ },
247
+ "failed_case_count": {
248
+ "type": "integer",
249
+ "minimum": 0,
250
+ "description": "v0.3.1 (optional). Number of cases marked failed_timeout (excluded from aggregates)."
251
+ },
252
+ "runner_version": {
253
+ "type": "string",
254
+ "minLength": 1
255
+ },
256
+ "date_utc": {
257
+ "type": "string",
258
+ "description": "ISO 8601 UTC timestamp of when the run finished.",
259
+ "pattern": "^\\d{4}-\\d{2}-\\d{2}T\\d{2}:\\d{2}:\\d{2}"
260
+ },
261
+ "registry": {
262
+ "type": "string",
263
+ "description": "Whether model_id resolved in the model registry (config/models.json). 'unregistered' means the run still executed but the model was unknown, so cost estimates used the conservative default price.",
264
+ "enum": [
265
+ "registered",
266
+ "unregistered"
267
+ ]
268
+ },
269
+ "transcripts": {
270
+ "type": "string",
271
+ "description": "Transcript retention for this run. 'hashes-only' = only the sha256 hashes in results.cases are kept (the default). 'retained-local' = the raw generations + judge outputs were also written to transcripts/<receipt-id>/ (gitignored by default). 'none' = nothing retained, not even hashes — ONLY honest on an imported (DECLARED) receipt.",
272
+ "enum": [
273
+ "retained-local",
274
+ "hashes-only",
275
+ "none"
276
+ ]
277
+ },
278
+ "judge": {
279
+ "type": "object",
280
+ "additionalProperties": false,
281
+ "description": "Judge sampling settings for this run.",
282
+ "required": [
283
+ "samples",
284
+ "temperature",
285
+ "sampling"
286
+ ],
287
+ "properties": {
288
+ "samples": {
289
+ "type": "integer",
290
+ "minimum": 1,
291
+ "description": "Judge samples taken per case."
292
+ },
293
+ "temperature": {
294
+ "type": [
295
+ "number",
296
+ "null"
297
+ ],
298
+ "description": "Judge temperature when the surface allows setting it (api -> 0), else null (cli -> surface-controlled)."
299
+ },
300
+ "sampling": {
301
+ "type": "string",
302
+ "description": "How sampling params were controlled, e.g. 'api-temperature-0' or 'surface-controlled'."
303
+ },
304
+ "surface": {
305
+ "type": "string",
306
+ "enum": [
307
+ "api",
308
+ "claude-cli",
309
+ "openai-api",
310
+ "openai-cli",
311
+ "external"
312
+ ]
313
+ }
314
+ }
315
+ }
316
+ }
317
+ },
318
+ "results": {
319
+ "type": "object",
320
+ "additionalProperties": false,
321
+ "required": [
322
+ "cases",
323
+ "aggregates"
324
+ ],
325
+ "properties": {
326
+ "cases": {
327
+ "type": "array",
328
+ "items": {
329
+ "type": "object",
330
+ "additionalProperties": false,
331
+ "required": [
332
+ "id",
333
+ "mode"
334
+ ],
335
+ "allOf": [
336
+ {
337
+ "description": "A completed case carries the full sampled band + hashes; a failed_timeout case is recorded WITHOUT fabricated samples (it is excluded from aggregates).",
338
+ "if": {
339
+ "required": [
340
+ "case_status"
341
+ ],
342
+ "properties": {
343
+ "case_status": {
344
+ "const": "failed_timeout"
345
+ }
346
+ }
347
+ },
348
+ "then": {
349
+ "required": [
350
+ "id",
351
+ "mode",
352
+ "case_status"
353
+ ]
354
+ },
355
+ "else": {
356
+ "required": [
357
+ "outcome",
358
+ "score",
359
+ "mean",
360
+ "stddev",
361
+ "samples",
362
+ "judge"
363
+ ]
364
+ }
365
+ }
366
+ ],
367
+ "properties": {
368
+ "id": {
369
+ "type": "string",
370
+ "minLength": 1
371
+ },
372
+ "mode": {
373
+ "type": "string",
374
+ "enum": [
375
+ "with_skill",
376
+ "baseline"
377
+ ]
378
+ },
379
+ "case_status": {
380
+ "type": "string",
381
+ "description": "v0.3.1 (optional; default 'ok'). 'failed_timeout' = the case's model/judge call persistently timed out after retries; the case is recorded but EXCLUDED from aggregates/verdicts — no samples/hashes are fabricated for it.",
382
+ "enum": [
383
+ "ok",
384
+ "failed_timeout"
385
+ ]
386
+ },
387
+ "outcome": {
388
+ "type": "string",
389
+ "description": "borderline = the threshold lies within mean +/- stddev.",
390
+ "enum": [
391
+ "pass",
392
+ "fail",
393
+ "borderline",
394
+ "score"
395
+ ]
396
+ },
397
+ "score": {
398
+ "type": "number",
399
+ "minimum": 0,
400
+ "maximum": 1,
401
+ "description": "Alias of mean, kept for v0.1 readers."
402
+ },
403
+ "mean": {
404
+ "type": "number",
405
+ "minimum": 0,
406
+ "maximum": 1
407
+ },
408
+ "stddev": {
409
+ "type": "number",
410
+ "minimum": 0,
411
+ "description": "Sample stddev of the judge samples (raw band half-width)."
412
+ },
413
+ "samples": {
414
+ "type": "array",
415
+ "items": {
416
+ "type": "number",
417
+ "minimum": 0,
418
+ "maximum": 1
419
+ },
420
+ "minItems": 1
421
+ },
422
+ "generation_hash": {
423
+ "type": "string",
424
+ "description": "sha256 (hex) of the raw model generation that was judged for this (case, mode).",
425
+ "pattern": "^[a-f0-9]{64}$"
426
+ },
427
+ "judge_sample_hashes": {
428
+ "type": "array",
429
+ "description": "sha256 (hex) of each raw judge output, one per judge sample. Same length as `samples`.",
430
+ "items": {
431
+ "type": "string",
432
+ "pattern": "^[a-f0-9]{64}$"
433
+ },
434
+ "minItems": 1
435
+ },
436
+ "threshold": {
437
+ "type": [
438
+ "number",
439
+ "null"
440
+ ],
441
+ "minimum": 0,
442
+ "maximum": 1
443
+ },
444
+ "reason": {
445
+ "type": "string"
446
+ },
447
+ "checks": {
448
+ "type": "array",
449
+ "description": "v0.3.1 (optional). Deterministic post-check results for this (case, mode): structural/regex assertions run on the model output ALONGSIDE the judge. Supplementary evidence reported as a separate column — NOT folded into the outcome/band verdict.",
450
+ "items": {
451
+ "type": "object",
452
+ "additionalProperties": false,
453
+ "required": [
454
+ "name",
455
+ "kind",
456
+ "pass"
457
+ ],
458
+ "properties": {
459
+ "name": {
460
+ "type": "string",
461
+ "minLength": 1
462
+ },
463
+ "kind": {
464
+ "type": "string",
465
+ "enum": [
466
+ "regex",
467
+ "contains",
468
+ "not_contains",
469
+ "min_length"
470
+ ]
471
+ },
472
+ "pass": {
473
+ "type": "boolean"
474
+ }
475
+ }
476
+ }
477
+ },
478
+ "judge": {
479
+ "type": "object",
480
+ "additionalProperties": false,
481
+ "required": [
482
+ "model_id",
483
+ "rubric_hash"
484
+ ],
485
+ "properties": {
486
+ "model_id": {
487
+ "type": "string",
488
+ "minLength": 1
489
+ },
490
+ "rubric_hash": {
491
+ "type": [
492
+ "string",
493
+ "null"
494
+ ],
495
+ "description": "Null ONLY on an imported (DECLARED) receipt whose rubric bytes were never seen.",
496
+ "pattern": "^[a-f0-9]{64}$"
497
+ }
498
+ }
499
+ }
500
+ }
501
+ }
502
+ },
503
+ "aggregates": {
504
+ "type": "object",
505
+ "additionalProperties": false,
506
+ "required": [
507
+ "with_skill",
508
+ "baseline"
509
+ ],
510
+ "properties": {
511
+ "with_skill": {
512
+ "$ref": "#/$defs/modeAggregate"
513
+ },
514
+ "baseline": {
515
+ "$ref": "#/$defs/modeAggregate"
516
+ }
517
+ }
518
+ }
519
+ }
520
+ },
521
+ "comparison": {
522
+ "type": "object",
523
+ "additionalProperties": false,
524
+ "required": [
525
+ "with_skill_score",
526
+ "baseline_score",
527
+ "delta",
528
+ "delta_uncertainty"
529
+ ],
530
+ "properties": {
531
+ "with_skill_score": {
532
+ "type": "number",
533
+ "minimum": 0,
534
+ "maximum": 1
535
+ },
536
+ "baseline_score": {
537
+ "type": [
538
+ "number",
539
+ "null"
540
+ ],
541
+ "minimum": 0,
542
+ "maximum": 1,
543
+ "description": "Null ONLY on an imported (DECLARED) receipt from a tool with no baseline mode — never a fabricated 0."
544
+ },
545
+ "delta": {
546
+ "type": [
547
+ "number",
548
+ "null"
549
+ ],
550
+ "minimum": -1,
551
+ "maximum": 1,
552
+ "description": "Null when baseline_score is null (no baseline mode was run)."
553
+ },
554
+ "delta_uncertainty": {
555
+ "type": [
556
+ "number",
557
+ "null"
558
+ ],
559
+ "minimum": 0,
560
+ "description": "Combined uncertainty of the delta (quadrature sum of the two aggregate bands). Null when delta is null."
561
+ }
562
+ }
563
+ },
564
+ "verification_level": {
565
+ "type": "string",
566
+ "description": "Community verification lattice. FORMAL is reserved/unimplemented in v0.3.",
567
+ "enum": [
568
+ "UNVERIFIED",
569
+ "DECLARED",
570
+ "TESTED"
571
+ ]
572
+ },
573
+ "editorial_reviews": {
574
+ "type": "array",
575
+ "description": "Optional pointers to external one-shot editorial reviews of this skill (context only; not verification evidence).",
576
+ "items": {
577
+ "type": "object",
578
+ "additionalProperties": false,
579
+ "required": [
580
+ "url",
581
+ "source",
582
+ "date"
583
+ ],
584
+ "properties": {
585
+ "url": {
586
+ "type": "string",
587
+ "minLength": 1
588
+ },
589
+ "source": {
590
+ "type": "string",
591
+ "minLength": 1
592
+ },
593
+ "date": {
594
+ "type": "string",
595
+ "pattern": "^\\d{4}-\\d{2}-\\d{2}$"
596
+ }
597
+ }
598
+ }
599
+ },
600
+ "receipt_hash": {
601
+ "type": "string",
602
+ "description": "sha256 (hex) of the canonical receipt JSON with this field omitted.",
603
+ "pattern": "^[a-f0-9]{64}$"
604
+ }
605
+ },
606
+ "$defs": {
607
+ "modeAggregate": {
608
+ "type": "object",
609
+ "additionalProperties": false,
610
+ "required": [
611
+ "case_count",
612
+ "pass_count",
613
+ "mean_score",
614
+ "stddev"
615
+ ],
616
+ "properties": {
617
+ "case_count": {
618
+ "type": "integer",
619
+ "minimum": 0
620
+ },
621
+ "pass_count": {
622
+ "type": "integer",
623
+ "minimum": 0
624
+ },
625
+ "borderline_count": {
626
+ "type": "integer",
627
+ "minimum": 0
628
+ },
629
+ "mean_score": {
630
+ "type": "number",
631
+ "minimum": 0,
632
+ "maximum": 1
633
+ },
634
+ "stddev": {
635
+ "type": "number",
636
+ "minimum": 0,
637
+ "description": "Suite dispersion: stddev of the per-case means across the suite. A reported summary stat; the drift headline is driven by per-case band-overlap verdicts, not this band."
638
+ }
639
+ }
640
+ }
641
+ }
642
+ }