driftproof 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +63 -19
- package/bin/driftproof +66 -4
- package/config/models.json +14 -4
- package/config.js +39 -5
- package/lib/checks.js +50 -0
- package/lib/diff.js +28 -7
- package/lib/export.js +52 -0
- package/lib/importers.js +207 -0
- package/lib/judge.js +27 -13
- package/lib/models.js +58 -8
- package/lib/provider.js +279 -36
- package/lib/receipt.js +22 -3
- package/lib/run.js +80 -26
- package/lib/skill.js +12 -2
- package/lib/skillCost.js +31 -0
- package/lib/verdict.js +10 -1
- package/package.json +1 -1
- package/spec/RECEIPT.md +72 -7
- package/spec/receipt.schema.json +141 -18
- package/spec/receipt.v0.3.schema.json +211 -0
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://driftproofhq.com/spec/receipt.schema.json",
|
|
4
|
+
"title": "driftproof receipt",
|
|
5
|
+
"description": "A signed, dated record of running one agent skill's eval suite with and without the skill on one model version, with sampled judge scores and confidence bands. Receipt spec v0.3 — adds transcript auditability (per-sample sha256 hashes + optional transcript retention) and registry/pricing provenance.",
|
|
6
|
+
"type": "object",
|
|
7
|
+
"additionalProperties": false,
|
|
8
|
+
"required": ["schema_version", "skill", "suite", "run", "results", "comparison", "verification_level", "receipt_hash"],
|
|
9
|
+
"properties": {
|
|
10
|
+
"schema_version": {
|
|
11
|
+
"type": "string",
|
|
12
|
+
"const": "0.3"
|
|
13
|
+
},
|
|
14
|
+
"skill": {
|
|
15
|
+
"type": "object",
|
|
16
|
+
"additionalProperties": false,
|
|
17
|
+
"required": ["name", "version", "content_hash"],
|
|
18
|
+
"properties": {
|
|
19
|
+
"name": { "type": "string", "minLength": 1 },
|
|
20
|
+
"version": { "type": "string", "minLength": 1 },
|
|
21
|
+
"content_hash": {
|
|
22
|
+
"type": "string",
|
|
23
|
+
"description": "sha256 (hex) over SKILL.md + all bundled files in canonical path-sorted order.",
|
|
24
|
+
"pattern": "^[a-f0-9]{64}$"
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
},
|
|
28
|
+
"suite": {
|
|
29
|
+
"type": "object",
|
|
30
|
+
"additionalProperties": false,
|
|
31
|
+
"required": ["format", "suite_hash", "case_count"],
|
|
32
|
+
"properties": {
|
|
33
|
+
"format": { "type": "string", "const": "agentskills.io/evals" },
|
|
34
|
+
"suite_hash": {
|
|
35
|
+
"type": "string",
|
|
36
|
+
"description": "sha256 (hex) over the canonicalized normalized case list.",
|
|
37
|
+
"pattern": "^[a-f0-9]{64}$"
|
|
38
|
+
},
|
|
39
|
+
"case_count": { "type": "integer", "minimum": 0 }
|
|
40
|
+
}
|
|
41
|
+
},
|
|
42
|
+
"run": {
|
|
43
|
+
"type": "object",
|
|
44
|
+
"additionalProperties": false,
|
|
45
|
+
"required": ["model_id", "surface", "runner_version", "date_utc", "judge", "registry", "transcripts"],
|
|
46
|
+
"properties": {
|
|
47
|
+
"model_id": { "type": "string", "minLength": 1 },
|
|
48
|
+
"model_release_date": {
|
|
49
|
+
"description": "ISO date (YYYY-MM-DD) of the model release if known, else null.",
|
|
50
|
+
"type": ["string", "null"],
|
|
51
|
+
"pattern": "^\\d{4}-\\d{2}-\\d{2}$"
|
|
52
|
+
},
|
|
53
|
+
"surface": { "type": "string", "enum": ["api", "claude-cli"] },
|
|
54
|
+
"runner_version": { "type": "string", "minLength": 1 },
|
|
55
|
+
"date_utc": {
|
|
56
|
+
"type": "string",
|
|
57
|
+
"description": "ISO 8601 UTC timestamp of when the run finished.",
|
|
58
|
+
"pattern": "^\\d{4}-\\d{2}-\\d{2}T\\d{2}:\\d{2}:\\d{2}"
|
|
59
|
+
},
|
|
60
|
+
"registry": {
|
|
61
|
+
"type": "string",
|
|
62
|
+
"description": "Whether model_id resolved in the model registry (config/models.json). 'unregistered' means the run still executed but the model was unknown, so cost estimates used the conservative default price.",
|
|
63
|
+
"enum": ["registered", "unregistered"]
|
|
64
|
+
},
|
|
65
|
+
"transcripts": {
|
|
66
|
+
"type": "string",
|
|
67
|
+
"description": "Transcript retention for this run. 'hashes-only' = only the sha256 hashes in results.cases are kept (the default). 'retained-local' = the raw generations + judge outputs were also written to transcripts/<receipt-id>/ (gitignored by default).",
|
|
68
|
+
"enum": ["retained-local", "hashes-only"]
|
|
69
|
+
},
|
|
70
|
+
"judge": {
|
|
71
|
+
"type": "object",
|
|
72
|
+
"additionalProperties": false,
|
|
73
|
+
"description": "Judge sampling settings for this run.",
|
|
74
|
+
"required": ["samples", "temperature", "sampling"],
|
|
75
|
+
"properties": {
|
|
76
|
+
"samples": { "type": "integer", "minimum": 1, "description": "Judge samples taken per case." },
|
|
77
|
+
"temperature": {
|
|
78
|
+
"type": ["number", "null"],
|
|
79
|
+
"description": "Judge temperature when the surface allows setting it (api -> 0), else null (cli -> surface-controlled)."
|
|
80
|
+
},
|
|
81
|
+
"sampling": {
|
|
82
|
+
"type": "string",
|
|
83
|
+
"description": "How sampling params were controlled, e.g. 'api-temperature-0' or 'surface-controlled'."
|
|
84
|
+
},
|
|
85
|
+
"surface": { "type": "string", "enum": ["api", "claude-cli"] }
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
},
|
|
90
|
+
"results": {
|
|
91
|
+
"type": "object",
|
|
92
|
+
"additionalProperties": false,
|
|
93
|
+
"required": ["cases", "aggregates"],
|
|
94
|
+
"properties": {
|
|
95
|
+
"cases": {
|
|
96
|
+
"type": "array",
|
|
97
|
+
"items": {
|
|
98
|
+
"type": "object",
|
|
99
|
+
"additionalProperties": false,
|
|
100
|
+
"required": ["id", "mode", "outcome", "score", "mean", "stddev", "samples", "judge", "generation_hash", "judge_sample_hashes"],
|
|
101
|
+
"properties": {
|
|
102
|
+
"id": { "type": "string", "minLength": 1 },
|
|
103
|
+
"mode": { "type": "string", "enum": ["with_skill", "baseline"] },
|
|
104
|
+
"outcome": {
|
|
105
|
+
"type": "string",
|
|
106
|
+
"description": "borderline = the threshold lies within mean +/- stddev.",
|
|
107
|
+
"enum": ["pass", "fail", "borderline", "score"]
|
|
108
|
+
},
|
|
109
|
+
"score": { "type": "number", "minimum": 0, "maximum": 1, "description": "Alias of mean, kept for v0.1 readers." },
|
|
110
|
+
"mean": { "type": "number", "minimum": 0, "maximum": 1 },
|
|
111
|
+
"stddev": { "type": "number", "minimum": 0, "description": "Sample stddev of the judge samples (raw band half-width)." },
|
|
112
|
+
"samples": {
|
|
113
|
+
"type": "array",
|
|
114
|
+
"items": { "type": "number", "minimum": 0, "maximum": 1 },
|
|
115
|
+
"minItems": 1
|
|
116
|
+
},
|
|
117
|
+
"generation_hash": {
|
|
118
|
+
"type": "string",
|
|
119
|
+
"description": "sha256 (hex) of the raw model generation that was judged for this (case, mode).",
|
|
120
|
+
"pattern": "^[a-f0-9]{64}$"
|
|
121
|
+
},
|
|
122
|
+
"judge_sample_hashes": {
|
|
123
|
+
"type": "array",
|
|
124
|
+
"description": "sha256 (hex) of each raw judge output, one per judge sample. Same length as `samples`.",
|
|
125
|
+
"items": { "type": "string", "pattern": "^[a-f0-9]{64}$" },
|
|
126
|
+
"minItems": 1
|
|
127
|
+
},
|
|
128
|
+
"threshold": { "type": ["number", "null"], "minimum": 0, "maximum": 1 },
|
|
129
|
+
"reason": { "type": "string" },
|
|
130
|
+
"judge": {
|
|
131
|
+
"type": "object",
|
|
132
|
+
"additionalProperties": false,
|
|
133
|
+
"required": ["model_id", "rubric_hash"],
|
|
134
|
+
"properties": {
|
|
135
|
+
"model_id": { "type": "string", "minLength": 1 },
|
|
136
|
+
"rubric_hash": { "type": "string", "pattern": "^[a-f0-9]{64}$" }
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
},
|
|
142
|
+
"aggregates": {
|
|
143
|
+
"type": "object",
|
|
144
|
+
"additionalProperties": false,
|
|
145
|
+
"required": ["with_skill", "baseline"],
|
|
146
|
+
"properties": {
|
|
147
|
+
"with_skill": { "$ref": "#/$defs/modeAggregate" },
|
|
148
|
+
"baseline": { "$ref": "#/$defs/modeAggregate" }
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
},
|
|
153
|
+
"comparison": {
|
|
154
|
+
"type": "object",
|
|
155
|
+
"additionalProperties": false,
|
|
156
|
+
"required": ["with_skill_score", "baseline_score", "delta", "delta_uncertainty"],
|
|
157
|
+
"properties": {
|
|
158
|
+
"with_skill_score": { "type": "number", "minimum": 0, "maximum": 1 },
|
|
159
|
+
"baseline_score": { "type": "number", "minimum": 0, "maximum": 1 },
|
|
160
|
+
"delta": { "type": "number", "minimum": -1, "maximum": 1 },
|
|
161
|
+
"delta_uncertainty": {
|
|
162
|
+
"type": "number",
|
|
163
|
+
"minimum": 0,
|
|
164
|
+
"description": "Combined uncertainty of the delta (quadrature sum of the two aggregate bands)."
|
|
165
|
+
}
|
|
166
|
+
}
|
|
167
|
+
},
|
|
168
|
+
"verification_level": {
|
|
169
|
+
"type": "string",
|
|
170
|
+
"description": "Community verification lattice. FORMAL is reserved/unimplemented in v0.3.",
|
|
171
|
+
"enum": ["UNVERIFIED", "DECLARED", "TESTED"]
|
|
172
|
+
},
|
|
173
|
+
"editorial_reviews": {
|
|
174
|
+
"type": "array",
|
|
175
|
+
"description": "Optional pointers to external one-shot editorial reviews of this skill (context only; not verification evidence).",
|
|
176
|
+
"items": {
|
|
177
|
+
"type": "object",
|
|
178
|
+
"additionalProperties": false,
|
|
179
|
+
"required": ["url", "source", "date"],
|
|
180
|
+
"properties": {
|
|
181
|
+
"url": { "type": "string", "minLength": 1 },
|
|
182
|
+
"source": { "type": "string", "minLength": 1 },
|
|
183
|
+
"date": { "type": "string", "pattern": "^\\d{4}-\\d{2}-\\d{2}$" }
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
},
|
|
187
|
+
"receipt_hash": {
|
|
188
|
+
"type": "string",
|
|
189
|
+
"description": "sha256 (hex) of the canonical receipt JSON with this field omitted.",
|
|
190
|
+
"pattern": "^[a-f0-9]{64}$"
|
|
191
|
+
}
|
|
192
|
+
},
|
|
193
|
+
"$defs": {
|
|
194
|
+
"modeAggregate": {
|
|
195
|
+
"type": "object",
|
|
196
|
+
"additionalProperties": false,
|
|
197
|
+
"required": ["case_count", "pass_count", "mean_score", "stddev"],
|
|
198
|
+
"properties": {
|
|
199
|
+
"case_count": { "type": "integer", "minimum": 0 },
|
|
200
|
+
"pass_count": { "type": "integer", "minimum": 0 },
|
|
201
|
+
"borderline_count": { "type": "integer", "minimum": 0 },
|
|
202
|
+
"mean_score": { "type": "number", "minimum": 0, "maximum": 1 },
|
|
203
|
+
"stddev": {
|
|
204
|
+
"type": "number",
|
|
205
|
+
"minimum": 0,
|
|
206
|
+
"description": "Suite dispersion: stddev of the per-case means across the suite. A reported summary stat; the drift headline is driven by per-case band-overlap verdicts, not this band."
|
|
207
|
+
}
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
}
|