driftproof 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +274 -0
- package/bin/driftproof +325 -0
- package/config/models.json +19 -0
- package/config.js +48 -0
- package/lib/canonical.js +41 -0
- package/lib/cost.js +118 -0
- package/lib/diff.js +152 -0
- package/lib/init.js +128 -0
- package/lib/json.js +141 -0
- package/lib/judge.js +126 -0
- package/lib/models.js +125 -0
- package/lib/provider.js +118 -0
- package/lib/receipt.js +133 -0
- package/lib/run.js +222 -0
- package/lib/runner.js +100 -0
- package/lib/skill.js +117 -0
- package/lib/stats.js +66 -0
- package/lib/stub.js +54 -0
- package/lib/verdict.js +76 -0
- package/package.json +45 -0
- package/spec/RECEIPT.md +246 -0
- package/spec/receipt.schema.json +211 -0
- package/spec/receipt.v0.1.schema.json +138 -0
- package/spec/receipt.v0.2.schema.json +190 -0
package/lib/skill.js
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
const fs = require('fs');
|
|
5
|
+
const path = require('path');
|
|
6
|
+
const { sha256Files, sha256Canonical } = require('./canonical');
|
|
7
|
+
const { SUITE_FORMAT } = require('../config');
|
|
8
|
+
|
|
9
|
+
// Load a skill directory and its eval suite.
|
|
10
|
+
//
|
|
11
|
+
// Expected layout (agentskills.io style):
|
|
12
|
+
// <skill-dir>/SKILL.md — the skill instructions (required)
|
|
13
|
+
// <skill-dir>/evals/evals.json — the eval suite (required)
|
|
14
|
+
// <skill-dir>/** — any other bundled files contribute to
|
|
15
|
+
// the content hash (references, scripts…)
|
|
16
|
+
//
|
|
17
|
+
// content_hash = sha256 over { SKILL.md, ...bundled files } in canonical (path-
|
|
18
|
+
// sorted) order, so the same skill on any machine yields the same hash and any
|
|
19
|
+
// edit to any bundled file changes it.
|
|
20
|
+
|
|
21
|
+
const IGNORE_DIRS = new Set(['.git', 'node_modules', 'evals']);
|
|
22
|
+
const IGNORE_FILES = new Set(['.DS_Store']);
|
|
23
|
+
|
|
24
|
+
function walkFiles(dir, base = dir, acc = []) {
|
|
25
|
+
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
|
26
|
+
if (entry.isDirectory()) {
|
|
27
|
+
if (IGNORE_DIRS.has(entry.name)) continue;
|
|
28
|
+
walkFiles(path.join(dir, entry.name), base, acc);
|
|
29
|
+
} else if (entry.isFile() && !IGNORE_FILES.has(entry.name)) {
|
|
30
|
+
const abs = path.join(dir, entry.name);
|
|
31
|
+
acc.push({ path: path.relative(base, abs), bytes: fs.readFileSync(abs) });
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
return acc;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
function loadSkill(skillDir) {
|
|
38
|
+
const dir = path.resolve(skillDir);
|
|
39
|
+
const skillMdPath = path.join(dir, 'SKILL.md');
|
|
40
|
+
if (!fs.existsSync(skillMdPath)) {
|
|
41
|
+
throw new Error(`no SKILL.md found in ${dir}`);
|
|
42
|
+
}
|
|
43
|
+
const skillMd = fs.readFileSync(skillMdPath, 'utf8');
|
|
44
|
+
|
|
45
|
+
// Content hash over SKILL.md + every bundled file (evals/ excluded — the suite
|
|
46
|
+
// is hashed separately so a suite edit doesn't masquerade as a skill change).
|
|
47
|
+
const files = walkFiles(dir);
|
|
48
|
+
const contentHash = sha256Files(files);
|
|
49
|
+
|
|
50
|
+
// Parse skill name/version from front-matter or the first H1; fall back to dir.
|
|
51
|
+
const meta = parseSkillMeta(skillMd);
|
|
52
|
+
const name = meta.name || path.basename(dir);
|
|
53
|
+
const version = meta.version || '0.0.0';
|
|
54
|
+
|
|
55
|
+
// Load the eval suite.
|
|
56
|
+
const suitePath = path.join(dir, 'evals', 'evals.json');
|
|
57
|
+
if (!fs.existsSync(suitePath)) {
|
|
58
|
+
throw new Error(`no evals/evals.json found in ${dir}`);
|
|
59
|
+
}
|
|
60
|
+
const suiteRaw = JSON.parse(fs.readFileSync(suitePath, 'utf8'));
|
|
61
|
+
const cases = normalizeCases(suiteRaw);
|
|
62
|
+
const suiteHash = sha256Canonical(cases);
|
|
63
|
+
|
|
64
|
+
return {
|
|
65
|
+
dir,
|
|
66
|
+
name,
|
|
67
|
+
version,
|
|
68
|
+
skillMd,
|
|
69
|
+
contentHash,
|
|
70
|
+
suite: {
|
|
71
|
+
format: SUITE_FORMAT,
|
|
72
|
+
suiteHash,
|
|
73
|
+
caseCount: cases.length,
|
|
74
|
+
cases,
|
|
75
|
+
},
|
|
76
|
+
};
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
// Pull name/version from a YAML-ish front-matter block, else from the H1 line.
|
|
80
|
+
function parseSkillMeta(md) {
|
|
81
|
+
const meta = {};
|
|
82
|
+
const fm = md.match(/^---\s*\n([\s\S]*?)\n---/);
|
|
83
|
+
if (fm) {
|
|
84
|
+
for (const line of fm[1].split('\n')) {
|
|
85
|
+
const m = line.match(/^\s*([A-Za-z_]+)\s*:\s*(.+?)\s*$/);
|
|
86
|
+
if (m) meta[m[1].toLowerCase()] = m[2].replace(/^["']|["']$/g, '');
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
if (!meta.name) {
|
|
90
|
+
const h1 = md.match(/^#\s+(.+)$/m);
|
|
91
|
+
if (h1) meta.name = h1[1].trim();
|
|
92
|
+
}
|
|
93
|
+
return meta;
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
// Accept a few shapes of the agentskills.io eval suite and normalize each case
|
|
97
|
+
// to { id, prompt, rubric, pass_threshold }. `cases` may live at the top level
|
|
98
|
+
// or under a `cases`/`evals` key.
|
|
99
|
+
function normalizeCases(raw) {
|
|
100
|
+
const list = Array.isArray(raw) ? raw
|
|
101
|
+
: Array.isArray(raw.cases) ? raw.cases
|
|
102
|
+
: Array.isArray(raw.evals) ? raw.evals
|
|
103
|
+
: null;
|
|
104
|
+
if (!list) throw new Error('evals.json must be an array or have a `cases`/`evals` array');
|
|
105
|
+
return list.map((c, i) => {
|
|
106
|
+
const id = String(c.id || c.name || `case-${i + 1}`);
|
|
107
|
+
const prompt = c.prompt || c.input || c.task;
|
|
108
|
+
const rubric = c.rubric || c.criteria || c.expected;
|
|
109
|
+
if (!prompt) throw new Error(`case "${id}" is missing a prompt/input/task`);
|
|
110
|
+
if (!rubric) throw new Error(`case "${id}" is missing a rubric/criteria/expected`);
|
|
111
|
+
const threshold = typeof c.pass_threshold === 'number' ? c.pass_threshold
|
|
112
|
+
: typeof c.threshold === 'number' ? c.threshold : 0.7;
|
|
113
|
+
return { id, prompt: String(prompt), rubric: String(rubric), pass_threshold: threshold };
|
|
114
|
+
});
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
module.exports = { loadSkill, normalizeCases, parseSkillMeta };
|
package/lib/stats.js
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// Small statistics helpers for sampled judging and confidence bands. Kept
|
|
5
|
+
// dependency-free and deterministic.
|
|
6
|
+
|
|
7
|
+
function round(n, dp = 6) { const f = Math.pow(10, dp); return Math.round(n * f) / f; }
|
|
8
|
+
|
|
9
|
+
function mean(xs) {
|
|
10
|
+
if (!xs.length) return 0;
|
|
11
|
+
return round(xs.reduce((a, b) => a + b, 0) / xs.length);
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
// Sample standard deviation (Bessel's n-1). Returns 0 for n < 2 (a single
|
|
15
|
+
// sample has no measurable spread). This is the RAW spread of a set of judge
|
|
16
|
+
// scores — used for the per-case ± band (borderline rule + per-case drift).
|
|
17
|
+
function stddev(xs) {
|
|
18
|
+
const n = xs.length;
|
|
19
|
+
if (n < 2) return 0;
|
|
20
|
+
const m = xs.reduce((a, b) => a + b, 0) / n;
|
|
21
|
+
const varr = xs.reduce((a, b) => a + (b - m) * (b - m), 0) / (n - 1);
|
|
22
|
+
return round(Math.sqrt(varr));
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
// Standard error of a mean: stddev / sqrt(n). This is the uncertainty OF THE
|
|
26
|
+
// MEAN (shrinks as you take more samples) and is the right band for aggregates,
|
|
27
|
+
// which is what run-to-run reproducibility of a headline number actually is.
|
|
28
|
+
function stderr(xs) {
|
|
29
|
+
const n = xs.length;
|
|
30
|
+
if (n < 2) return 0;
|
|
31
|
+
return round(stddev(xs) / Math.sqrt(n));
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
// Combine independent uncertainties in quadrature: sqrt(a^2 + b^2).
|
|
35
|
+
function combineUncertainty(a, b) {
|
|
36
|
+
return round(Math.sqrt(a * a + b * b));
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
// Aggregate a set of per-case measurements, each {mean, stddev, n}, into an
|
|
40
|
+
// overall mean and a band that is the DISPERSION of the per-case means across
|
|
41
|
+
// the suite (sample stddev of the case means) — NOT the standard error of the
|
|
42
|
+
// mean. Rationale: the standard error shrinks with sampling/case-count and makes
|
|
43
|
+
// the headline hypersensitive (a trivial move reads as a real change), which is
|
|
44
|
+
// exactly the cry-wolf failure this project exists to prevent. Suite dispersion
|
|
45
|
+
// is a conventional, honest "mean ± stddev across the suite". The drift HEADLINE
|
|
46
|
+
// verdict is driven by the per-case band-overlap verdicts (see lib/diff.js), not
|
|
47
|
+
// by this aggregate band; this value is a reported summary statistic.
|
|
48
|
+
function aggregateBands(cases) {
|
|
49
|
+
if (!cases.length) return { mean: 0, stddev: 0 };
|
|
50
|
+
const means = cases.map((c) => c.mean);
|
|
51
|
+
return { mean: mean(means), stddev: stddev(means) };
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
// Do two confidence bands (mean ± half-width) fail to overlap, and in which
|
|
55
|
+
// direction? Returns 'regression' (b below a), 'improvement' (b above a), or
|
|
56
|
+
// 'within noise' (bands touch/overlap). This is the anti-false-positive rule:
|
|
57
|
+
// a change is only claimed when the bands are fully separated.
|
|
58
|
+
// regression : meanB + hwB < meanA - hwA
|
|
59
|
+
// improvement : meanB - hwB > meanA + hwA
|
|
60
|
+
function bandVerdict(meanA, hwA, meanB, hwB) {
|
|
61
|
+
if (meanB + hwB < meanA - hwA) return 'regression';
|
|
62
|
+
if (meanB - hwB > meanA + hwA) return 'improvement';
|
|
63
|
+
return 'within noise';
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
module.exports = { mean, stddev, stderr, combineUncertainty, aggregateBands, bandVerdict, round };
|
package/lib/stub.js
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// Offline stub surface (DRIFTPROOF_STUB=1).
|
|
5
|
+
//
|
|
6
|
+
// When set, provider.complete() delegates here and NEVER contacts a model — no
|
|
7
|
+
// API call, no `claude` spawn, zero spend. It exists so the GitHub Action (and
|
|
8
|
+
// CI generally) can be exercised end-to-end for free: the runner still builds a
|
|
9
|
+
// fully-formed, schema-valid, self-hashed receipt with a real with-skill-vs-
|
|
10
|
+
// baseline lift; only the underlying text is canned.
|
|
11
|
+
//
|
|
12
|
+
// Determinism is the whole point, so scores carry no variance (stddev 0). The
|
|
13
|
+
// stub is model- and skill-agnostic: it distinguishes the two things the runner
|
|
14
|
+
// actually varies — whether the SKILL.md was supplied (with_skill vs baseline) —
|
|
15
|
+
// by marking each canned generation, and the canned judge grades on that marker.
|
|
16
|
+
|
|
17
|
+
// Markers embedded in canned generations. The judge stub reads them back out of
|
|
18
|
+
// the response it is handed, which is how a zero-model-call run still produces a
|
|
19
|
+
// non-trivial (positive) skill lift.
|
|
20
|
+
const MARK_SKILL = 'DRIFTPROOF_STUB_S'; // generation produced WITH the skill
|
|
21
|
+
const MARK_BASE = 'DRIFTPROOF_STUB_B'; // baseline generation (no skill)
|
|
22
|
+
|
|
23
|
+
// A judge call is recognised structurally (no import of judge.js — that would be
|
|
24
|
+
// circular, since judge.js requires provider.js): the judge prompt always ends by
|
|
25
|
+
// asking for a strict JSON object.
|
|
26
|
+
function isJudgePrompt(prompt) {
|
|
27
|
+
return /Return ONLY this JSON object/.test(String(prompt || ''));
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
// Deterministic canned completion. `system` truthy ⇒ a with-skill generation
|
|
31
|
+
// (the runner passes SKILL.md as the system prompt only in with_skill mode).
|
|
32
|
+
function stubComplete({ system, prompt }) {
|
|
33
|
+
if (isJudgePrompt(prompt)) {
|
|
34
|
+
// Grade the response we were handed: reward the with-skill marker.
|
|
35
|
+
const helped = new RegExp(MARK_SKILL).test(String(prompt || ''));
|
|
36
|
+
const score = helped ? 0.85 : 0.42;
|
|
37
|
+
const body = {
|
|
38
|
+
score,
|
|
39
|
+
pass: score >= 0.7,
|
|
40
|
+
reason: `stub judge (DRIFTPROOF_STUB): ${helped ? 'with-skill marker present' : 'baseline'}`,
|
|
41
|
+
};
|
|
42
|
+
return { text: JSON.stringify(body), usage: null };
|
|
43
|
+
}
|
|
44
|
+
// Generation call: canned, marked by mode.
|
|
45
|
+
const mark = system ? MARK_SKILL : MARK_BASE;
|
|
46
|
+
const text = `${mark}\nfeat(stub): canned offline generation for CI (no model was called)`;
|
|
47
|
+
return { text, usage: null };
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
function stubEnabled() {
|
|
51
|
+
return process.env.DRIFTPROOF_STUB === '1';
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
module.exports = { stubComplete, stubEnabled, MARK_SKILL, MARK_BASE };
|
package/lib/verdict.js
ADDED
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
const { EFFECT_FLOOR } = require('../config');
|
|
5
|
+
|
|
6
|
+
// Single-receipt verdict + shields.io badge.
|
|
7
|
+
//
|
|
8
|
+
// `diff` compares TWO receipts across model releases; a single receipt already
|
|
9
|
+
// carries its own honest verdict: does the skill still help on THIS model? That
|
|
10
|
+
// is exactly `comparison.delta` (with-skill mean minus baseline mean), read
|
|
11
|
+
// through the SAME practical-significance floor the drift rule uses (config.js
|
|
12
|
+
// EFFECT_FLOOR). Band separation is not applied here — a single run has no second
|
|
13
|
+
// band to separate from — but the effect floor is, so a trivial lift below the
|
|
14
|
+
// judge's quantization grid is reported as "no effect", never as "passing".
|
|
15
|
+
//
|
|
16
|
+
// delta >= EFFECT_FLOOR → PASSED (the skill measurably helps)
|
|
17
|
+
// |delta| < EFFECT_FLOOR → NO_EFFECT (within the judge's noise floor)
|
|
18
|
+
// delta <= -EFFECT_FLOOR → REGRESSED (the skill measurably hurts on this model)
|
|
19
|
+
|
|
20
|
+
const VERDICTS = {
|
|
21
|
+
PASSED: { color: 'brightgreen', word: 'passing' },
|
|
22
|
+
NO_EFFECT: { color: 'lightgrey', word: 'no effect' },
|
|
23
|
+
REGRESSED: { color: 'red', word: 'regressed' },
|
|
24
|
+
};
|
|
25
|
+
|
|
26
|
+
// Strip a trailing -YYYYMMDD date stamp so the badge reads cleanly
|
|
27
|
+
// (claude-haiku-4-5-20251001 → claude-haiku-4-5).
|
|
28
|
+
function shortModel(modelId) {
|
|
29
|
+
return String(modelId || 'unknown').replace(/-\d{8}$/, '');
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
// Derive the verdict object from a receipt. Returns
|
|
33
|
+
// { verdict, delta, floor, model, message, color }
|
|
34
|
+
function verdictFromReceipt(receipt) {
|
|
35
|
+
const cmp = (receipt && receipt.comparison) || {};
|
|
36
|
+
const delta = typeof cmp.delta === 'number' ? cmp.delta : 0;
|
|
37
|
+
const model = shortModel(receipt && receipt.run && receipt.run.model_id);
|
|
38
|
+
let verdict;
|
|
39
|
+
if (delta >= EFFECT_FLOOR) verdict = 'PASSED';
|
|
40
|
+
else if (delta <= -EFFECT_FLOOR) verdict = 'REGRESSED';
|
|
41
|
+
else verdict = 'NO_EFFECT';
|
|
42
|
+
const meta = VERDICTS[verdict];
|
|
43
|
+
return {
|
|
44
|
+
verdict,
|
|
45
|
+
delta,
|
|
46
|
+
floor: EFFECT_FLOOR,
|
|
47
|
+
model,
|
|
48
|
+
message: `${meta.word} on ${model}`,
|
|
49
|
+
color: meta.color,
|
|
50
|
+
};
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
// A shields.io "endpoint" badge object (schemaVersion 1). Committed as JSON and
|
|
54
|
+
// referenced via https://img.shields.io/endpoint?url=<public-url-of-this-json>.
|
|
55
|
+
function badgeEndpoint(receipt, { label = 'driftproof' } = {}) {
|
|
56
|
+
const v = verdictFromReceipt(receipt);
|
|
57
|
+
return {
|
|
58
|
+
schemaVersion: 1,
|
|
59
|
+
label,
|
|
60
|
+
message: v.message,
|
|
61
|
+
color: v.color,
|
|
62
|
+
};
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
// Lines suitable for appending to a GitHub Actions $GITHUB_OUTPUT file.
|
|
66
|
+
function githubOutputLines(receipt) {
|
|
67
|
+
const v = verdictFromReceipt(receipt);
|
|
68
|
+
return [
|
|
69
|
+
`verdict=${v.verdict}`,
|
|
70
|
+
`delta=${v.delta}`,
|
|
71
|
+
`message=${v.message}`,
|
|
72
|
+
`color=${v.color}`,
|
|
73
|
+
].join('\n');
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
module.exports = { verdictFromReceipt, badgeEndpoint, githubOutputLines, shortModel, VERDICTS };
|
package/package.json
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "driftproof",
|
|
3
|
+
"version": "0.3.0",
|
|
4
|
+
"description": "Continuous verification of agent skills: run a skill's eval suite with and without the skill across model versions, emit signed dated receipts, and diff receipts into drift reports.",
|
|
5
|
+
"license": "Apache-2.0",
|
|
6
|
+
"keywords": [
|
|
7
|
+
"agent-skills",
|
|
8
|
+
"evals",
|
|
9
|
+
"llm",
|
|
10
|
+
"drift",
|
|
11
|
+
"verification"
|
|
12
|
+
],
|
|
13
|
+
"homepage": "https://driftproofhq.com",
|
|
14
|
+
"repository": {
|
|
15
|
+
"type": "git",
|
|
16
|
+
"url": "git+https://github.com/driftproofhq/driftproof.git"
|
|
17
|
+
},
|
|
18
|
+
"bugs": {
|
|
19
|
+
"url": "https://github.com/driftproofhq/driftproof/issues"
|
|
20
|
+
},
|
|
21
|
+
"bin": {
|
|
22
|
+
"driftproof": "bin/driftproof"
|
|
23
|
+
},
|
|
24
|
+
"files": [
|
|
25
|
+
"bin/",
|
|
26
|
+
"lib/",
|
|
27
|
+
"spec/",
|
|
28
|
+
"config.js",
|
|
29
|
+
"config/models.json"
|
|
30
|
+
],
|
|
31
|
+
"scripts": {
|
|
32
|
+
"gate": "node tests/gate.js",
|
|
33
|
+
"run": "node bin/driftproof run",
|
|
34
|
+
"diff": "node bin/driftproof diff"
|
|
35
|
+
},
|
|
36
|
+
"dependencies": {
|
|
37
|
+
"ajv": "^8.17.1"
|
|
38
|
+
},
|
|
39
|
+
"optionalDependencies": {
|
|
40
|
+
"@anthropic-ai/sdk": "^0.39.0"
|
|
41
|
+
},
|
|
42
|
+
"engines": {
|
|
43
|
+
"node": ">=22"
|
|
44
|
+
}
|
|
45
|
+
}
|
package/spec/RECEIPT.md
ADDED
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
<!-- SPDX-License-Identifier: Apache-2.0 -->
|
|
2
|
+
# Driftproof receipt — spec v0.3
|
|
3
|
+
|
|
4
|
+
A **receipt** is a signed, dated record of running one agent skill's eval suite
|
|
5
|
+
**with** and **without** the skill on one model version, with the judge **sampled**
|
|
6
|
+
so every score carries a confidence band. Receipts are the unit of evidence
|
|
7
|
+
Driftproof produces; diffing two receipts across model releases yields a **drift
|
|
8
|
+
report**.
|
|
9
|
+
|
|
10
|
+
The machine-readable contract is [`receipt.schema.json`](./receipt.schema.json)
|
|
11
|
+
(JSON Schema, draft 2020-12). This document is the human companion. Where they
|
|
12
|
+
disagree, the schema wins.
|
|
13
|
+
|
|
14
|
+
**Versioning.** The current schema is v0.3
|
|
15
|
+
([`receipt.schema.json`](./receipt.schema.json)). Prior schemas are kept as
|
|
16
|
+
[`receipt.v0.2.schema.json`](./receipt.v0.2.schema.json) and
|
|
17
|
+
[`receipt.v0.1.schema.json`](./receipt.v0.1.schema.json); the validator picks the
|
|
18
|
+
schema by the receipt's own `schema_version`, so v0.1 and v0.2 receipts still load
|
|
19
|
+
and validate. v0.3 is an **additive** bump — every field it adds is required in a
|
|
20
|
+
fresh run, but older receipts are read unchanged against their own schema.
|
|
21
|
+
|
|
22
|
+
## Design goals
|
|
23
|
+
|
|
24
|
+
- **Reproducible.** Every hash is over a *canonical* JSON form (object keys sorted
|
|
25
|
+
at every level, no insignificant whitespace), so the same inputs produce the
|
|
26
|
+
same hashes on any machine.
|
|
27
|
+
- **Credible.** An LLM judge is noisy. v0.2 samples the judge N times per case and
|
|
28
|
+
records the distribution, so a receipt carries a *band*, not a single fragile
|
|
29
|
+
number — and drift reports only claim a regression when bands don't overlap.
|
|
30
|
+
- **Tamper-evident (lite).** The `receipt_hash` is a self-hash: recompute over the
|
|
31
|
+
receipt with `receipt_hash` removed and compare. A hand-edit breaks it. This is
|
|
32
|
+
integrity, **not** authenticity — v0.3 does not sign with a key (see open
|
|
33
|
+
questions; "signed" is a promissory note until then).
|
|
34
|
+
- **Consumes, doesn't invent.** The eval suite format is `agentskills.io/evals`.
|
|
35
|
+
|
|
36
|
+
## What changed from v0.2 (transcript auditability + provenance)
|
|
37
|
+
|
|
38
|
+
- **Per-case `generation_hash`** — sha256 (hex) of the raw model generation that
|
|
39
|
+
was judged for that (case, mode). Binds the graded case to the exact text.
|
|
40
|
+
- **Per-case `judge_sample_hashes[]`** — sha256 (hex) of each raw judge output,
|
|
41
|
+
one per judge sample (same length as `samples`). A reader can check every score
|
|
42
|
+
a verdict rests on against a retained transcript.
|
|
43
|
+
- **`run.transcripts`** — `"retained-local"` when the raw generations + judge
|
|
44
|
+
outputs were also written to `transcripts/<receipt-id>/` (under
|
|
45
|
+
`--keep-transcripts`), else `"hashes-only"` (the default — only the hashes).
|
|
46
|
+
- **`run.registry`** — `"registered"` when `model_id` resolved in the model
|
|
47
|
+
registry (`config/models.json`), else `"unregistered"` (the run still executed;
|
|
48
|
+
cost was estimated with a conservative default price).
|
|
49
|
+
- These four fields are **required** in a v0.3 receipt. Everything else is
|
|
50
|
+
unchanged from v0.2.
|
|
51
|
+
|
|
52
|
+
## What changed from v0.1
|
|
53
|
+
|
|
54
|
+
- Per-case: added `mean`, `stddev`, `samples[]`; `outcome` gains **`borderline`**.
|
|
55
|
+
- `run`: added a required `judge` block (samples, temperature, sampling, surface).
|
|
56
|
+
- Aggregates: added `stddev` (the run-to-run band) and `borderline_count`.
|
|
57
|
+
- `comparison`: added `delta_uncertainty`.
|
|
58
|
+
- Added optional top-level `editorial_reviews[]`.
|
|
59
|
+
- `score` is retained as an alias of `mean` so v0.1 readers keep working.
|
|
60
|
+
|
|
61
|
+
## Fields
|
|
62
|
+
|
|
63
|
+
### `schema_version` (string, required) — `"0.3"`.
|
|
64
|
+
|
|
65
|
+
### `skill` (object, required)
|
|
66
|
+
| field | type | notes |
|
|
67
|
+
|---|---|---|
|
|
68
|
+
| `name` | string | From SKILL.md front-matter or its H1. |
|
|
69
|
+
| `version` | string | Skill's declared version. |
|
|
70
|
+
| `content_hash` | sha256 hex | Over `SKILL.md` + every bundled file (the `evals/` dir is **excluded** — hashed separately as `suite_hash`), path + content, path-sorted. |
|
|
71
|
+
|
|
72
|
+
### `suite` (object, required)
|
|
73
|
+
| field | type | notes |
|
|
74
|
+
|---|---|---|
|
|
75
|
+
| `format` | string | Always `"agentskills.io/evals"`. |
|
|
76
|
+
| `suite_hash` | sha256 hex | Over the canonicalized normalized case list (`{id, prompt, rubric, pass_threshold}`). |
|
|
77
|
+
| `case_count` | integer | Cases in the suite (before any `--max-cases` cap). |
|
|
78
|
+
|
|
79
|
+
### `run` (object, required)
|
|
80
|
+
| field | type | notes |
|
|
81
|
+
|---|---|---|
|
|
82
|
+
| `model_id` | string | Canonical model id run on. |
|
|
83
|
+
| `model_release_date` | ISO date or `null` | Release date **if known**, else `null`. |
|
|
84
|
+
| `surface` | `"api"` \| `"claude-cli"` | `api` = Messages API; `claude-cli` = spawned `claude -p` (subscription). |
|
|
85
|
+
| `runner_version` | string | Runner version, for reproducibility. |
|
|
86
|
+
| `date_utc` | ISO 8601 UTC | When the run finished. |
|
|
87
|
+
| `registry` | `"registered"` \| `"unregistered"` | **v0.3.** Whether `model_id` resolved in the model registry (`config/models.json`). |
|
|
88
|
+
| `transcripts` | `"retained-local"` \| `"hashes-only"` | **v0.3.** Whether the raw generations + judge outputs were retained on disk (see § Transcripts). |
|
|
89
|
+
| `judge` | object | `{ samples, temperature, sampling, surface }` — see below. |
|
|
90
|
+
|
|
91
|
+
**`run.judge`** records how grading was done: `samples` (judge calls per case),
|
|
92
|
+
`temperature` (a number when the surface lets us set it — the `api` surface pins
|
|
93
|
+
**0** — else `null`), `sampling` (`"api-temperature-0"` or `"surface-controlled"`),
|
|
94
|
+
and `surface`. Determinism where the surface allows: on `api` the judge is pinned
|
|
95
|
+
to temperature 0; on `claude-cli` sampling params are surface-controlled and the
|
|
96
|
+
receipt says so.
|
|
97
|
+
|
|
98
|
+
### `results` (object, required)
|
|
99
|
+
- **`cases`** — one entry **per (case, mode)**; every case appears twice
|
|
100
|
+
(`with_skill` and `baseline`):
|
|
101
|
+
| field | type | notes |
|
|
102
|
+
|---|---|---|
|
|
103
|
+
| `id` | string | Eval case id. |
|
|
104
|
+
| `mode` | `"with_skill"` \| `"baseline"` | Whether SKILL.md was supplied. |
|
|
105
|
+
| `outcome` | `"pass"` \| `"fail"` \| `"borderline"` \| `"score"` | **`borderline`** = the threshold lies within `mean ± stddev` (the run can't confidently call it). `score` = un-thresholded case. |
|
|
106
|
+
| `mean` | number 0–1 | Mean of the judge samples. |
|
|
107
|
+
| `stddev` | number ≥ 0 | Sample stddev of the judge samples — the raw per-case band half-width used by the borderline rule and per-case drift. |
|
|
108
|
+
| `samples` | number[] | The individual judge scores. |
|
|
109
|
+
| `generation_hash` | sha256 hex | **v0.3.** sha256 of the raw model generation that was judged for this (case, mode). |
|
|
110
|
+
| `judge_sample_hashes` | sha256 hex[] | **v0.3.** sha256 of each raw judge output, one per sample (same length as `samples`). |
|
|
111
|
+
| `score` | number 0–1 | Alias of `mean` (kept for v0.1 readers). |
|
|
112
|
+
| `threshold` | number or `null` | The case's pass threshold (or `null`). |
|
|
113
|
+
| `reason` | string | One-line judge rationale (optional). |
|
|
114
|
+
| `judge` | object | `{ model_id, rubric_hash }` — who graded and a hash binding the grade to the exact rubric + judge system prompt. |
|
|
115
|
+
- **`aggregates`** — `{ with_skill, baseline }`, each
|
|
116
|
+
`{ case_count, pass_count, borderline_count?, mean_score, stddev }`. Here
|
|
117
|
+
`stddev` is the **suite dispersion**: the stddev of the per-case means across
|
|
118
|
+
the suite (a conventional "mean ± stddev across items"). It is a reported
|
|
119
|
+
summary stat — the drift **headline verdict is driven by the per-case
|
|
120
|
+
band-overlap verdicts**, not by a separate test on this aggregate band. (Using
|
|
121
|
+
the standard error of the mean here instead would shrink with sampling and make
|
|
122
|
+
the headline cry wolf on trivial moves — precisely what this design avoids.)
|
|
123
|
+
|
|
124
|
+
### `comparison` (object, required)
|
|
125
|
+
| field | type | notes |
|
|
126
|
+
|---|---|---|
|
|
127
|
+
| `with_skill_score` | number 0–1 | Aggregate with-skill mean. |
|
|
128
|
+
| `baseline_score` | number 0–1 | Aggregate baseline mean. |
|
|
129
|
+
| `delta` | number −1–1 | The skill's measured lift. |
|
|
130
|
+
| `delta_uncertainty` | number ≥ 0 | Combined band on the delta (quadrature sum of the two aggregate bands). |
|
|
131
|
+
|
|
132
|
+
### `verification_level` (string, required)
|
|
133
|
+
Community lattice: **UNVERIFIED** (bare claim) · **DECLARED** (author asserts, no
|
|
134
|
+
run) · **TESTED** (a suite was executed and judged — what Driftproof emits) ·
|
|
135
|
+
**FORMAL** *(reserved / unimplemented; the schema rejects it)*.
|
|
136
|
+
|
|
137
|
+
### `editorial_reviews` (array, optional)
|
|
138
|
+
Pointers to external one-shot reviews of the skill, each `{ url, source, date }`.
|
|
139
|
+
**Context only — not verification evidence.** A one-shot editorial verdict and a
|
|
140
|
+
dated, model-bound receipt are different things; this field lets a receipt link
|
|
141
|
+
the former without conflating it with the latter.
|
|
142
|
+
|
|
143
|
+
### `receipt_hash` (sha256 hex, required)
|
|
144
|
+
sha256 over the canonical receipt JSON **with `receipt_hash` removed**.
|
|
145
|
+
|
|
146
|
+
## Transcripts (v0.3)
|
|
147
|
+
|
|
148
|
+
Every generation and every judge sample is hashed into the receipt
|
|
149
|
+
(`generation_hash`, `judge_sample_hashes[]`), so a verdict is always checkable
|
|
150
|
+
*in principle*. `--keep-transcripts` makes it checkable *in practice*: the raw
|
|
151
|
+
generations and judge outputs are written to `transcripts/<receipt_hash>/` (one
|
|
152
|
+
JSON per case+mode, plus an `index.json`), and the receipt records
|
|
153
|
+
`transcripts: "retained-local"`. Without the flag the receipt records
|
|
154
|
+
`transcripts: "hashes-only"` and only the hashes are kept. The transcript
|
|
155
|
+
directory is **gitignored by default** — raw model text is never committed. The
|
|
156
|
+
default for trigger-initiated runs is `retained-local` (disk is cheap, audits are
|
|
157
|
+
not). To verify a retained receipt: re-hash each transcript file and compare
|
|
158
|
+
against the receipt's `generation_hash` / `judge_sample_hashes`.
|
|
159
|
+
|
|
160
|
+
## Drift verdict (how `diff` reads two receipts)
|
|
161
|
+
|
|
162
|
+
For each case, Driftproof compares the two `with_skill` bands (`mean ± stddev`).
|
|
163
|
+
A verdict requires **two** conditions — band separation **and** a minimum effect:
|
|
164
|
+
|
|
165
|
+
1. **Band separation** (geometry):
|
|
166
|
+
- **regression** — `meanB + stddevB < meanA − stddevA` (B's band is entirely below A's)
|
|
167
|
+
- **improvement** — `meanB − stddevB > meanA + stddevA`
|
|
168
|
+
- **within noise** — otherwise (the bands touch or overlap)
|
|
169
|
+
2. **Minimum effect floor** — `|meanB − meanA| ≥ EFFECT_FLOOR` (default **0.05**, see
|
|
170
|
+
`config.js`). A regression/improvement from step 1 whose mean moved by *less*
|
|
171
|
+
than the floor is downgraded to **`within noise (below effect floor)`**.
|
|
172
|
+
|
|
173
|
+
**Why the floor.** The LLM judge quantizes scores to a coarse ~0.05–0.1 grid, so a
|
|
174
|
+
confident grade frequently has `stddev 0` — a zero-width "point band." Two point
|
|
175
|
+
bands one quantum apart (e.g. `0.60` vs `0.64`) are technically non-overlapping yet
|
|
176
|
+
represent no meaningful behaviour change. The floor stops that statistically-real-
|
|
177
|
+
but-practically-trivial move from being reported as drift. `0.05` ≈ one judge
|
|
178
|
+
quantization step; the comparison is inclusive (a move of exactly 0.05 counts).
|
|
179
|
+
|
|
180
|
+
A regression is claimed **only** when both conditions hold. The **headline verdict
|
|
181
|
+
summarizes the per-case verdicts** (e.g. "DRIFT — 2 cases regressed", or "WITHIN
|
|
182
|
+
NOISE" when none moved beyond its band + floor); it does not run a separate, tighter
|
|
183
|
+
test on the aggregate mean. This is the anti-false-positive core: band separation
|
|
184
|
+
**plus a real-sized delta**, not either alone, is what triggers a verdict.
|
|
185
|
+
|
|
186
|
+
## Example (abridged)
|
|
187
|
+
|
|
188
|
+
```json
|
|
189
|
+
{
|
|
190
|
+
"schema_version": "0.3",
|
|
191
|
+
"skill": { "name": "commit-message-conventions", "version": "0.2.0", "content_hash": "…64 hex…" },
|
|
192
|
+
"suite": { "format": "agentskills.io/evals", "suite_hash": "…64 hex…", "case_count": 10 },
|
|
193
|
+
"run": {
|
|
194
|
+
"model_id": "claude-haiku-4-5-20251001", "model_release_date": "2025-10-01",
|
|
195
|
+
"surface": "claude-cli", "runner_version": "0.3.0", "date_utc": "2026-07-27T09:15:00.000Z",
|
|
196
|
+
"registry": "registered", "transcripts": "hashes-only",
|
|
197
|
+
"judge": { "samples": 5, "temperature": null, "sampling": "surface-controlled", "surface": "claude-cli" }
|
|
198
|
+
},
|
|
199
|
+
"results": {
|
|
200
|
+
"cases": [
|
|
201
|
+
{ "id": "perf-not-refactor", "mode": "with_skill", "outcome": "pass",
|
|
202
|
+
"score": 0.86, "mean": 0.86, "stddev": 0.05, "samples": [0.9,0.8,0.85,0.9,0.85],
|
|
203
|
+
"generation_hash": "…64 hex…", "judge_sample_hashes": ["…64 hex…","…","…","…","…"],
|
|
204
|
+
"threshold": 0.7, "judge": { "model_id": "claude-haiku-4-5-20251001", "rubric_hash": "…64 hex…" } }
|
|
205
|
+
],
|
|
206
|
+
"aggregates": {
|
|
207
|
+
"with_skill": { "case_count": 10, "pass_count": 7, "borderline_count": 1, "mean_score": 0.81, "stddev": 0.02 },
|
|
208
|
+
"baseline": { "case_count": 10, "pass_count": 1, "mean_score": 0.42, "stddev": 0.03 }
|
|
209
|
+
}
|
|
210
|
+
},
|
|
211
|
+
"comparison": { "with_skill_score": 0.81, "baseline_score": 0.42, "delta": 0.39, "delta_uncertainty": 0.036 },
|
|
212
|
+
"verification_level": "TESTED",
|
|
213
|
+
"receipt_hash": "…64 hex…"
|
|
214
|
+
}
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
## Open questions (v0.3)
|
|
218
|
+
|
|
219
|
+
1. **"Signed" is really "self-hashed".** `receipt_hash` is integrity, not
|
|
220
|
+
authenticity — anyone can recompute it after an edit. A key signature (Ed25519
|
|
221
|
+
over the canonical form, or in-toto/sigstore attestation) is the natural next
|
|
222
|
+
step and the main thing standing between v0.2 and a receipt a third party can
|
|
223
|
+
*trust*, not just *read*.
|
|
224
|
+
2. **Sampling captures judge variance, not generation variance.** v0.2 samples the
|
|
225
|
+
judge N times on a **single** generated response per (case, mode). That
|
|
226
|
+
quantifies judge noise — the dominant, cheapest-to-measure source — but a fresh
|
|
227
|
+
generation each run would vary too. Bands may therefore be tighter than true
|
|
228
|
+
run-to-run spread. Sampling generations (and separating the two variance
|
|
229
|
+
components) is a candidate for v0.3.
|
|
230
|
+
3. **Two different "bands" coexist.** Per-case `stddev` is judge-sample spread
|
|
231
|
+
(drives borderline + per-case drift); aggregate `stddev` is suite dispersion
|
|
232
|
+
(a summary stat). Neither is a true run-to-run confidence interval — building
|
|
233
|
+
one would need repeated full runs (generation + judge), which v0.2 does not do
|
|
234
|
+
(see #2). The headline sidesteps this by summarizing per-case verdicts rather
|
|
235
|
+
than testing the aggregate band.
|
|
236
|
+
4. **`model_release_date` provenance is unverified** (small built-in table + a date
|
|
237
|
+
parsed from the model id). Should cite a source or be omitted.
|
|
238
|
+
5. **`outcome: "score"` overlaps the numeric `score`/`mean`.** Mild redundancy,
|
|
239
|
+
kept so thresholded and un-thresholded suites both round-trip.
|
|
240
|
+
6. **No environment fingerprint** beyond `runner_version` — fine while the runner
|
|
241
|
+
is the only producer; matters once third parties emit receipts.
|
|
242
|
+
7. **Baseline construction is fixed** ("same prompt, no SKILL.md"). Skills that
|
|
243
|
+
assume tools/context a bare baseline never had can show inflated lift; a future
|
|
244
|
+
spec may need to declare the baseline condition.
|
|
245
|
+
8. **`editorial_reviews` scope.** Deliberately marked context-only; if it starts
|
|
246
|
+
being treated as evidence, it will need its own trust rules.
|