driftproof 0.11.2 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,7 +2,7 @@
2
2
  "$schema": "https://json-schema.org/draft/2020-12/schema",
3
3
  "$id": "https://driftproofhq.com/spec/receipt.schema.json",
4
4
  "title": "driftproof receipt",
5
- "description": "A hash-verified, dated record of running one agent skill eval suite with and without the skill on one model version, with the GENERATION sampled n times per arm and the judge sampled k times inside each draw. Receipt spec v0.7: over v0.6, the fields are unchanged and the verdict a reader derives gains UNDERPOWERED: a case the band rule does not separate whose spreads and measured draws cannot resolve a shift of the effect floor (spec 035, spec/RECEIPT.md § What v0.7 adds). Over v0.5, as v0.6 said, the receipt says what answered it: `run.answered_by` (kind model|stub|external, whether the surface echoed a model id and which, and which spawn path produced it); `run.surface` and `run.judge.surface` may read `stub` (nothing answered; the text was canned) and a TESTED receipt requires answered_by.kind model; `run.judge.model_id` and `run.judge.prompt_template_hash` say which judge ran; every draw records its `stop_reason` and whether it was `truncated` (a truncated draw is unmeasured, never judged) and the case counts `n_truncated`; `case_status` gains `failed_unmeasured` (every draw of the arm was unmeasured for a reason that is not a timeout), recorded without fabricated samples or hashes exactly as a timeout is; `results.aggregates.band_rule` states the formula the aggregate band is derived by; an aggregate `stddev` is null only when its arm has fewer than two cases and `comparison.delta_uncertainty` is null only beside `delta_uncertainty_unavailable`. Every prior receipt still validates against its own frozen schema (v0.1, v0.2, v0.3, v0.3.1, v0.4, v0.5, v0.6).",
5
+ "description": "Receipt spec v0.9: over v0.8, imported receipts say what produced them (spec 049, spec/RECEIPT.md § What v0.9 adds). Every field it adds is optional: `run.harness` (the agent harness that produced the answers), `run.import` (the imported document: tool, format, format version, sha256, the import time, any sidecar file read, and a notice for each thing the source did not establish), `skill.unit` (a skill, or a plugin measured whole), and per case `activation` (whether the skill fired, never a score), `excluded_draws` (runs kept out of the band, with their reasons) and `label` (the source's display name). `economics.basis` adds `source-list-price-estimate` and `source-reported`, an arm may carry `mean_total_tokens`, `run.provider` may be `unknown`, and `run.date_utc` may be null on an external receipt whose source records no date. v0.8 is frozen as receipt.v0.8.schema.json. Over v0.7, v0.8 said: counts and clocks (spec 043, spec/RECEIPT.md § What v0.8 adds). `run.judge.samples` is optional, so an absent count stays absent; generation and judge counts are separate optional fields, `generations_per_arm` and `judge_samples_per_generation`, held in a `counts` object at the narrowest honest scope (`run.counts`, `run.arms.<mode>.counts`, `results.cases[].counts`); `run.arms.<mode>` may name an archived arm's own `model_id` and `generated_at`; `run.generated_at` is when generation began, and `run.judged_at` with `run.grader_revision` record a re-judge and appear together or not at all; a case may be `no_observations`. v0.7 is frozen as receipt.v0.7.schema.json. A hash-verified, dated record of running one agent skill eval suite with and without the skill on one model version, with the GENERATION sampled n times per arm and the judge sampled k times inside each draw. Over v0.6, v0.7 said: the fields are unchanged and the verdict a reader derives gains UNDERPOWERED: a case the band rule does not separate whose spreads and measured draws cannot resolve a shift of the effect floor (spec 035, spec/RECEIPT.md § What v0.7 adds). Over v0.5, as v0.6 said, the receipt says what answered it: `run.answered_by` (kind model|stub|external, whether the surface echoed a model id and which, and which spawn path produced it); `run.surface` and `run.judge.surface` may read `stub` (nothing answered; the text was canned) and a TESTED receipt requires answered_by.kind model; `run.judge.model_id` and `run.judge.prompt_template_hash` say which judge ran; every draw records its `stop_reason` and whether it was `truncated` (a truncated draw is unmeasured, never judged) and the case counts `n_truncated`; `case_status` gains `failed_unmeasured` (every draw of the arm was unmeasured for a reason that is not a timeout), recorded without fabricated samples or hashes exactly as a timeout is; `results.aggregates.band_rule` states the formula the aggregate band is derived by; an aggregate `stddev` is null only when its arm has fewer than two cases and `comparison.delta_uncertainty` is null only beside `delta_uncertainty_unavailable`. Every prior receipt still validates against its own frozen schema (v0.1, v0.2, v0.3, v0.3.1, v0.4, v0.5, v0.6).",
6
6
  "type": "object",
7
7
  "additionalProperties": false,
8
8
  "required": [
@@ -274,7 +274,7 @@
274
274
  "properties": {
275
275
  "schema_version": {
276
276
  "type": "string",
277
- "const": "0.7"
277
+ "const": "0.9"
278
278
  },
279
279
  "skill": {
280
280
  "type": "object",
@@ -305,6 +305,14 @@
305
305
  "type": "integer",
306
306
  "minimum": 0,
307
307
  "description": "v0.3.1 (optional). Estimated token size of the skill's SKILL.md (a coarse chars/4 proxy, not a model tokenizer), used for the value-per-token axis (delta per 1k skill tokens). See docs/methodology.html."
308
+ },
309
+ "unit": {
310
+ "type": "string",
311
+ "enum": [
312
+ "skill",
313
+ "plugin"
314
+ ],
315
+ "description": "v0.9 (optional). What was measured: one skill, or a plugin loaded whole with every skill it carries (a claude plugin eval import). Absent means a skill."
308
316
  }
309
317
  }
310
318
  },
@@ -370,10 +378,11 @@
370
378
  },
371
379
  "provider": {
372
380
  "type": "string",
373
- "description": "v0.3.1. The two-axis provider the target model ran on (registry `provider`, else inferred from the id).",
381
+ "description": "v0.3.1. The two-axis provider the target model ran on (registry `provider`, else inferred from the id). v0.9: `unknown` on an imported receipt whose source records no model.",
374
382
  "enum": [
375
383
  "anthropic",
376
- "openai"
384
+ "openai",
385
+ "unknown"
377
386
  ]
378
387
  },
379
388
  "surface": {
@@ -415,8 +424,11 @@
415
424
  "minLength": 1
416
425
  },
417
426
  "date_utc": {
418
- "type": "string",
419
- "description": "ISO 8601 UTC timestamp of when the run finished.",
427
+ "type": [
428
+ "string",
429
+ "null"
430
+ ],
431
+ "description": "ISO 8601 UTC. As lib/run.js writes it: the time the run's calls finished when the CLI stamps it, or the stamp a caller passes, which for a batch is one stamp shared by every receipt in the batch. For when generation began, read run.generated_at (v0.8). Receipts before v0.8 described this field as when the run finished; that was true of the CLI and not of a batch. v0.9: null only on an external (imported) receipt whose source records no date; an import never writes the time it ran here (that is run.import.imported_at).",
420
432
  "pattern": "^\\d{4}-\\d{2}-\\d{2}T\\d{2}:\\d{2}:\\d{2}"
421
433
  },
422
434
  "registry": {
@@ -441,7 +453,6 @@
441
453
  "additionalProperties": false,
442
454
  "description": "Judge sampling settings for this run.",
443
455
  "required": [
444
- "samples",
445
456
  "temperature",
446
457
  "sampling",
447
458
  "model_id",
@@ -451,7 +462,7 @@
451
462
  "samples": {
452
463
  "type": "integer",
453
464
  "minimum": 1,
454
- "description": "Judge samples taken per case."
465
+ "description": "Judge samples taken per generation. v0.8: optional; absent means the count is unknown, never 1. When present it equals run.counts.judge_samples_per_generation if that is present (the loader refuses a receipt where they differ)."
455
466
  },
456
467
  "temperature": {
457
468
  "type": [
@@ -587,8 +598,214 @@
587
598
  ]
588
599
  }
589
600
  }
601
+ },
602
+ "generated_at": {
603
+ "type": "string",
604
+ "pattern": "^\\d{4}-\\d{2}-\\d{2}T\\d{2}:\\d{2}:\\d{2}",
605
+ "description": "v0.8 (optional). When generation began for this receipt, stamped by the runner before its first generation call. An archived arm's own generation time is in run.arms.<mode>.generated_at."
606
+ },
607
+ "judged_at": {
608
+ "type": "string",
609
+ "pattern": "^\\d{4}-\\d{2}-\\d{2}T\\d{2}:\\d{2}:\\d{2}",
610
+ "description": "v0.8 (optional). Present only on a re-judge: when frozen outputs were graded again. Requires grader_revision."
611
+ },
612
+ "grader_revision": {
613
+ "type": "object",
614
+ "additionalProperties": false,
615
+ "required": [
616
+ "prompt_template_hash",
617
+ "rubric_hashes"
618
+ ],
619
+ "description": "v0.8 (optional). Present only on a re-judge, with judged_at: the grading definition that re-judge applied.",
620
+ "properties": {
621
+ "prompt_template_hash": {
622
+ "type": [
623
+ "string",
624
+ "null"
625
+ ]
626
+ },
627
+ "rubric_hashes": {
628
+ "type": "array",
629
+ "items": {
630
+ "type": [
631
+ "string",
632
+ "null"
633
+ ]
634
+ }
635
+ }
636
+ }
637
+ },
638
+ "counts": {
639
+ "$ref": "#/$defs/counts"
640
+ },
641
+ "arms": {
642
+ "type": "object",
643
+ "additionalProperties": false,
644
+ "properties": {
645
+ "with_skill": {
646
+ "type": "object",
647
+ "additionalProperties": false,
648
+ "description": "v0.8 (optional). A per-arm override. An arm carrying its own generated_at is archived: it was generated in an earlier run, and it never inherits run.counts; an absent model_id or generated_at inherits the run's.",
649
+ "properties": {
650
+ "model_id": {
651
+ "type": "string",
652
+ "minLength": 1
653
+ },
654
+ "generated_at": {
655
+ "type": "string",
656
+ "pattern": "^\\d{4}-\\d{2}-\\d{2}T\\d{2}:\\d{2}:\\d{2}"
657
+ },
658
+ "counts": {
659
+ "$ref": "#/$defs/counts"
660
+ }
661
+ }
662
+ },
663
+ "baseline": {
664
+ "type": "object",
665
+ "additionalProperties": false,
666
+ "description": "v0.8 (optional). A per-arm override. An arm carrying its own generated_at is archived: it was generated in an earlier run, and it never inherits run.counts; an absent model_id or generated_at inherits the run's.",
667
+ "properties": {
668
+ "model_id": {
669
+ "type": "string",
670
+ "minLength": 1
671
+ },
672
+ "generated_at": {
673
+ "type": "string",
674
+ "pattern": "^\\d{4}-\\d{2}-\\d{2}T\\d{2}:\\d{2}:\\d{2}"
675
+ },
676
+ "counts": {
677
+ "$ref": "#/$defs/counts"
678
+ }
679
+ }
680
+ }
681
+ }
682
+ },
683
+ "harness": {
684
+ "type": "object",
685
+ "additionalProperties": false,
686
+ "required": [
687
+ "name",
688
+ "version"
689
+ ],
690
+ "description": "v0.9 (optional). The agent harness that produced the answers, as the source names it (e.g. claude-code and its version for a claude plugin eval import). version is null when the source does not record it.",
691
+ "properties": {
692
+ "name": {
693
+ "type": "string",
694
+ "minLength": 1
695
+ },
696
+ "version": {
697
+ "type": [
698
+ "string",
699
+ "null"
700
+ ]
701
+ }
702
+ }
703
+ },
704
+ "import": {
705
+ "type": "object",
706
+ "additionalProperties": false,
707
+ "required": [
708
+ "tool",
709
+ "format",
710
+ "format_version",
711
+ "source_sha256",
712
+ "imported_at"
713
+ ],
714
+ "description": "v0.9 (optional). The document an imported receipt was converted from. imported_at is the only place the import time appears.",
715
+ "properties": {
716
+ "tool": {
717
+ "type": "string",
718
+ "minLength": 1,
719
+ "description": "The --from value, e.g. claude-plugin-eval or skill-creator."
720
+ },
721
+ "format": {
722
+ "type": "string",
723
+ "minLength": 1,
724
+ "description": "The document read, e.g. aggregate-result.json or benchmark.json."
725
+ },
726
+ "format_version": {
727
+ "type": [
728
+ "string",
729
+ "integer",
730
+ "null"
731
+ ],
732
+ "description": "The document's own version field as written (claude plugin eval's schemaVersion), or null when the format carries none."
733
+ },
734
+ "source_sha256": {
735
+ "type": "string",
736
+ "pattern": "^[a-f0-9]{64}$",
737
+ "description": "sha256 of the document's bytes."
738
+ },
739
+ "imported_at": {
740
+ "type": "string",
741
+ "pattern": "^\\d{4}-\\d{2}-\\d{2}T\\d{2}:\\d{2}:\\d{2}",
742
+ "description": "ISO 8601 UTC: when the import ran. Never the run date."
743
+ },
744
+ "sidecars": {
745
+ "type": "array",
746
+ "description": "Other files the import read beside the document (skill-up's result.json), by name and sha256.",
747
+ "items": {
748
+ "type": "object",
749
+ "additionalProperties": false,
750
+ "required": [
751
+ "file",
752
+ "sha256"
753
+ ],
754
+ "properties": {
755
+ "file": {
756
+ "type": "string",
757
+ "minLength": 1
758
+ },
759
+ "sha256": {
760
+ "type": "string",
761
+ "pattern": "^[a-f0-9]{64}$"
762
+ }
763
+ }
764
+ }
765
+ },
766
+ "notices": {
767
+ "type": "array",
768
+ "description": "One sentence per thing the source did not establish, or per choice the import made that a reader must know (an unknown model, a date read from a directory name, errored runs counted).",
769
+ "items": {
770
+ "type": "string",
771
+ "minLength": 1
772
+ }
773
+ }
774
+ }
590
775
  }
591
- }
776
+ },
777
+ "dependentRequired": {
778
+ "judged_at": [
779
+ "grader_revision"
780
+ ],
781
+ "grader_revision": [
782
+ "judged_at"
783
+ ]
784
+ },
785
+ "allOf": [
786
+ {
787
+ "description": "v0.9: run.date_utc may be null only on an external (imported) receipt.",
788
+ "if": {
789
+ "not": {
790
+ "properties": {
791
+ "surface": {
792
+ "const": "external"
793
+ }
794
+ },
795
+ "required": [
796
+ "surface"
797
+ ]
798
+ }
799
+ },
800
+ "then": {
801
+ "properties": {
802
+ "date_utc": {
803
+ "type": "string"
804
+ }
805
+ }
806
+ }
807
+ }
808
+ ]
592
809
  },
593
810
  "results": {
594
811
  "type": "object",
@@ -636,7 +853,12 @@
636
853
  "stddev",
637
854
  "samples",
638
855
  "judge"
639
- ]
856
+ ],
857
+ "properties": {
858
+ "samples": {
859
+ "minItems": 1
860
+ }
861
+ }
640
862
  },
641
863
  "else": {
642
864
  "required": [
@@ -661,11 +883,12 @@
661
883
  },
662
884
  "case_status": {
663
885
  "type": "string",
664
- "description": "v0.3.1 (optional; default 'ok'). 'failed_timeout' = the case's model/judge call persistently timed out after retries. v0.6: 'failed_unmeasured' = every draw of the arm was unmeasured for a reason that is not a timeout (an empty generation, a judge output with no score, a truncated draw). Either way the case is recorded but EXCLUDED from aggregates/verdicts — no samples/hashes are fabricated for it.",
886
+ "description": "v0.3.1 (optional; default 'ok'). 'failed_timeout' = the case's model/judge call persistently timed out after retries. v0.6: 'failed_unmeasured' = every draw of the arm was unmeasured for a reason that is not a timeout (an empty generation, a judge output with no score, a truncated draw). Either way the case is recorded but EXCLUDED from aggregates/verdicts; no samples/hashes are fabricated for it. v0.8: 'no_observations' = the source supplied no observation for the case (an imported task whose trials list is empty, samples [], or absent, no samples); recorded, excluded from aggregates, named in excluded_cases.",
665
887
  "enum": [
666
888
  "ok",
667
889
  "failed_timeout",
668
- "failed_unmeasured"
890
+ "failed_unmeasured",
891
+ "no_observations"
669
892
  ]
670
893
  },
671
894
  "outcome": {
@@ -701,7 +924,7 @@
701
924
  "minimum": 0,
702
925
  "maximum": 1
703
926
  },
704
- "minItems": 1
927
+ "minItems": 0
705
928
  },
706
929
  "generation_hash": {
707
930
  "type": "string",
@@ -1108,6 +1331,65 @@
1108
1331
  }
1109
1332
  }
1110
1333
  ]
1334
+ },
1335
+ "counts": {
1336
+ "$ref": "#/$defs/counts"
1337
+ },
1338
+ "label": {
1339
+ "type": "string",
1340
+ "minLength": 1,
1341
+ "description": "v0.9 (optional). The source's display name for the case (benchmark.json eval_name). Cases are identified by id, never by label."
1342
+ },
1343
+ "activation": {
1344
+ "type": "array",
1345
+ "minItems": 1,
1346
+ "description": "v0.9 (optional). Whether the skill fired, per unscored skill-invocation indicator the source ran (a claude plugin eval tool_used: Skill grader with scored: false): fired of runs, over the measured draws of this case and arm. Never part of any score, band or comparison.",
1347
+ "items": {
1348
+ "type": "object",
1349
+ "additionalProperties": false,
1350
+ "required": [
1351
+ "indicator",
1352
+ "fired",
1353
+ "runs"
1354
+ ],
1355
+ "properties": {
1356
+ "indicator": {
1357
+ "type": "string",
1358
+ "minLength": 1
1359
+ },
1360
+ "fired": {
1361
+ "type": "integer",
1362
+ "minimum": 0
1363
+ },
1364
+ "runs": {
1365
+ "type": "integer",
1366
+ "minimum": 1
1367
+ }
1368
+ }
1369
+ }
1370
+ },
1371
+ "excluded_draws": {
1372
+ "type": "array",
1373
+ "minItems": 1,
1374
+ "description": "v0.9 (optional). Runs the source recorded for this case and arm that are kept out of its samples, each with its 1-based position in the source's list (or its run number) and the reason.",
1375
+ "items": {
1376
+ "type": "object",
1377
+ "additionalProperties": false,
1378
+ "required": [
1379
+ "draw_index",
1380
+ "reason"
1381
+ ],
1382
+ "properties": {
1383
+ "draw_index": {
1384
+ "type": "integer",
1385
+ "minimum": 1
1386
+ },
1387
+ "reason": {
1388
+ "type": "string",
1389
+ "minLength": 1
1390
+ }
1391
+ }
1392
+ }
1111
1393
  }
1112
1394
  }
1113
1395
  }
@@ -1295,9 +1577,11 @@
1295
1577
  "basis": {
1296
1578
  "enum": [
1297
1579
  "metered",
1298
- "metered-equivalent"
1580
+ "metered-equivalent",
1581
+ "source-list-price-estimate",
1582
+ "source-reported"
1299
1583
  ],
1300
- "description": "metered = real spend on an api surface; metered-equivalent = what the same tokens would have cost on the metered API (subscription CLI surfaces, where actual spend is $0)."
1584
+ "description": "metered = real spend on an api surface; metered-equivalent = what the same tokens would have cost on the metered API (subscription CLI surfaces, where actual spend is $0). v0.9, imported receipts only: source-list-price-estimate = the source tool's own cost estimate at list price, not a Driftproof figure; source-reported = tokens and time as the source reports them, with no cost."
1301
1585
  },
1302
1586
  "surface": {
1303
1587
  "type": "string"
@@ -1351,6 +1635,13 @@
1351
1635
  "number",
1352
1636
  "null"
1353
1637
  ]
1638
+ },
1639
+ "mean_total_tokens": {
1640
+ "type": [
1641
+ "number",
1642
+ "null"
1643
+ ],
1644
+ "description": "v0.9 (optional). Mean tokens per call where the source reports one total (benchmark.json result.tokens), not input and output apart."
1354
1645
  }
1355
1646
  }
1356
1647
  },
@@ -1403,6 +1694,13 @@
1403
1694
  "number",
1404
1695
  "null"
1405
1696
  ]
1697
+ },
1698
+ "mean_total_tokens": {
1699
+ "type": [
1700
+ "number",
1701
+ "null"
1702
+ ],
1703
+ "description": "v0.9 (optional). Mean tokens per call where the source reports one total (benchmark.json result.tokens), not input and output apart."
1406
1704
  }
1407
1705
  }
1408
1706
  },
@@ -1558,6 +1856,23 @@
1558
1856
  }
1559
1857
  }
1560
1858
  ]
1859
+ },
1860
+ "counts": {
1861
+ "type": "object",
1862
+ "additionalProperties": false,
1863
+ "description": "v0.8. How the data came to be, at the scope it holds: every case of every arm generated in this run (run.counts), one arm (run.arms.<mode>.counts), or one case and arm (results.cases[].counts). A reader takes the narrowest scope present. An absent field is unknown, never 1.",
1864
+ "properties": {
1865
+ "generations_per_arm": {
1866
+ "type": "integer",
1867
+ "minimum": 0,
1868
+ "description": "Independent candidate generations. 0 only where the source establishes zero."
1869
+ },
1870
+ "judge_samples_per_generation": {
1871
+ "type": "integer",
1872
+ "minimum": 1,
1873
+ "description": "Judge samples taken of each generation."
1874
+ }
1875
+ }
1561
1876
  }
1562
1877
  }
1563
1878
  }