driftproof 0.3.0 β 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +63 -19
- package/bin/driftproof +66 -4
- package/config/models.json +14 -4
- package/config.js +55 -5
- package/lib/checks.js +50 -0
- package/lib/diff.js +28 -7
- package/lib/export.js +52 -0
- package/lib/importers.js +207 -0
- package/lib/judge.js +35 -13
- package/lib/models.js +58 -8
- package/lib/provider.js +318 -45
- package/lib/receipt.js +29 -4
- package/lib/run.js +108 -27
- package/lib/skill.js +12 -2
- package/lib/skillCost.js +31 -0
- package/lib/stub.js +24 -3
- package/lib/usage.js +168 -0
- package/lib/value.js +502 -0
- package/lib/verdict.js +10 -1
- package/package.json +1 -1
- package/spec/RECEIPT.md +116 -8
- package/spec/receipt.schema.json +809 -59
- package/spec/receipt.v0.3.1.schema.json +642 -0
- package/spec/receipt.v0.3.schema.json +211 -0
package/README.md
CHANGED
|
@@ -15,9 +15,26 @@ across model releases and you get a **drift report**.
|
|
|
15
15
|
Driftproof **consumes** the [`agentskills.io/evals`](https://agentskills.io) eval
|
|
16
16
|
format; it does not invent its own.
|
|
17
17
|
|
|
18
|
-
π **
|
|
19
|
-
|
|
20
|
-
|
|
18
|
+
π **Four published reports** (each re-derived from committed receipts, nothing
|
|
19
|
+
hand-entered), spanning three report types that share one band-based, floor-gated
|
|
20
|
+
verdict rule and differ only in what moves underneath the skill:
|
|
21
|
+
|
|
22
|
+
- **[Report #001](https://driftproofhq.com/reports/001/)** β *release drift*: ten
|
|
23
|
+
public agent skills across a current-vs-previous Sonnet release; 9 of 10 moved
|
|
24
|
+
beyond noise. ([markdown](reports/report-001.md))
|
|
25
|
+
- **[Report #002](https://driftproofhq.com/reports/002/)** β *substrate
|
|
26
|
+
durability*: the same suites across two vendors' CLIs (Claude vs Codex);
|
|
27
|
+
3 durable, 4 substrate-dependent, 2 regressed, 1 no effect.
|
|
28
|
+
- **[Report #003](https://driftproofhq.com/reports/003/)** β *release drift*:
|
|
29
|
+
`claude-opus-4-8` β `claude-opus-5`; 4 improved, 2 regressed, 4 within noise.
|
|
30
|
+
- **[Report #004](https://driftproofhq.com/reports/004/)** β *capability gap*:
|
|
31
|
+
`claude-opus-5` (flagship) vs `claude-fable-5` (frontier tier);
|
|
32
|
+
3 durable, 5 tier-dependent, 0 regressions, 2 no effect β encoded expertise
|
|
33
|
+
survives the frontier tier.
|
|
34
|
+
|
|
35
|
+
βοΈ The launch essay, **[Three model releases later: what actually happens to agent
|
|
36
|
+
skills](https://driftproofhq.com/writing/three-releases/)**, reads the first three
|
|
37
|
+
reports together.
|
|
21
38
|
|
|
22
39
|
## Why
|
|
23
40
|
|
|
@@ -92,8 +109,21 @@ npx driftproof validate receipt.json
|
|
|
92
109
|
|
|
93
110
|
# Emit a shields.io badge from a receipt (see "Badge", below)
|
|
94
111
|
npx driftproof badge receipt.json --out badges/my-skill.json
|
|
112
|
+
|
|
113
|
+
# Interop: convert another tool's results into a DECLARED receipt (honest
|
|
114
|
+
# epistemics β no fabricated hashes, excluded from drift verdicts), and emit
|
|
115
|
+
# the minimal stable summary other tools can consume. See docs/interop.md.
|
|
116
|
+
npx driftproof import results.json --from agent-skills-eval # or: skillgrade
|
|
117
|
+
npx driftproof export receipt.json --to summary-json
|
|
95
118
|
```
|
|
96
119
|
|
|
120
|
+
Receipts are an **open format** β the JSON Schema is served at its canonical id
|
|
121
|
+
([driftproofhq.com/spec/receipt.schema.json](https://driftproofhq.com/spec/receipt.schema.json)),
|
|
122
|
+
and any harness is encouraged to emit them. The interop contract (DECLARED vs
|
|
123
|
+
TESTED, import mappings, `summary-json`) is documented in
|
|
124
|
+
[`docs/interop.md`](docs/interop.md) and on
|
|
125
|
+
[the interop page](https://driftproofhq.com/interop.html).
|
|
126
|
+
|
|
97
127
|
A skill directory is expected to look like:
|
|
98
128
|
|
|
99
129
|
```
|
|
@@ -132,16 +162,19 @@ A receipt is the unit of evidence β one JSON document conforming to
|
|
|
132
162
|
|
|
133
163
|
```jsonc
|
|
134
164
|
{
|
|
135
|
-
"schema_version": "0.
|
|
165
|
+
"schema_version": "0.4",
|
|
136
166
|
"skill": { "name": "commit-message-conventions", "version": "0.2.0",
|
|
137
167
|
"content_hash": "β¦sha256 over SKILL.md + bundled filesβ¦" },
|
|
138
168
|
"suite": { "format": "agentskills.io/evals", "suite_hash": "β¦", "case_count": 10 },
|
|
139
169
|
"run": {
|
|
140
170
|
"model_id": "claude-haiku-4-5-20251001",
|
|
141
171
|
"model_release_date": "2025-10-01",
|
|
172
|
+
"provider": "anthropic",
|
|
142
173
|
"surface": "claude-cli",
|
|
143
|
-
"runner_version": "0.
|
|
174
|
+
"runner_version": "0.5.0",
|
|
144
175
|
"date_utc": "2026-07-27Tβ¦Z",
|
|
176
|
+
"registry": "registered",
|
|
177
|
+
"transcripts": "hashes-only",
|
|
145
178
|
"judge": { "samples": 5, "temperature": null, "sampling": "surface-controlled",
|
|
146
179
|
"surface": "claude-cli" }
|
|
147
180
|
},
|
|
@@ -200,7 +233,7 @@ jobs:
|
|
|
200
233
|
runs-on: ubuntu-latest
|
|
201
234
|
steps:
|
|
202
235
|
- uses: actions/checkout@v4
|
|
203
|
-
- uses: driftproofhq/driftproof@v0.
|
|
236
|
+
- uses: driftproofhq/driftproof@v0.5.0
|
|
204
237
|
with:
|
|
205
238
|
skill-dir: skills/my-skill
|
|
206
239
|
models: claude-haiku-4-5
|
|
@@ -232,38 +265,49 @@ git add badges/my-skill.json && git commit -m "chore: driftproof badge"
|
|
|
232
265
|

|
|
233
266
|
```
|
|
234
267
|
|
|
235
|
-
The [badge at the top of this README](badges/)
|
|
268
|
+
The [badge at the top of this README](docs/badges/commit-message-conventions.json)
|
|
269
|
+
is the living demo: it is generated
|
|
236
270
|
from the `commit-message-conventions` example's own receipt and served from the
|
|
237
271
|
site, so it reflects a real dated run, not a hand-set color.
|
|
238
272
|
|
|
239
273
|
## Reports
|
|
240
274
|
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
275
|
+
Four reports are published, spanning three report types (see the roll at the top
|
|
276
|
+
of this README, and [REPORT-STYLE.md](REPORT-STYLE.md) for the shared chrome
|
|
277
|
+
every report inherits). Each report and every verdict in it are **re-derived from
|
|
278
|
+
the receipts** committed under [`receipts/`](receipts/)
|
|
279
|
+
(`receipts/report-001/` β¦ `receipts/report-004/`) β nothing is hand-entered.
|
|
245
280
|
|
|
246
281
|
Driftproof does **not** commit third-party skill content. Each `SKILL.md` is
|
|
247
282
|
fetched at run time from a pinned commit and verified by sha256 against
|
|
248
283
|
[`suites/manifest.json`](suites/manifest.json); we commit only the manifest, our
|
|
249
|
-
authored eval suites ([`suites/`](suites/)), the receipts, and the
|
|
250
|
-
Reproduce
|
|
284
|
+
authored eval suites ([`suites/`](suites/)), the receipts, and the reports.
|
|
285
|
+
Reproduce Report #001 in three commands:
|
|
251
286
|
|
|
252
287
|
```bash
|
|
253
288
|
node scripts/fetch-skills.js # fetch pinned SKILL.md files (sha256-verified) β untracked workdir
|
|
254
|
-
node scripts/run-report-001.js --concurrency 5 # run both models Γ with/baseline, emit
|
|
289
|
+
node scripts/run-report-001.js --concurrency 5 # run both models Γ with/baseline, emit receipts
|
|
255
290
|
node scripts/build-report-001.js # re-derive the report from the receipts
|
|
256
291
|
```
|
|
257
292
|
|
|
293
|
+
Reports #002β#004 have their own runners
|
|
294
|
+
(`scripts/prepare-report-00N.js`) following the same
|
|
295
|
+
fetch β run β re-derive shape.
|
|
296
|
+
|
|
297
|
+
**Model-release triggers are live**: `scripts/release-watch.js` (keyless β it
|
|
298
|
+
reads the public models registry) notices a new model release, re-runs the
|
|
299
|
+
receipts, and queues a draft report in `reports/pending-publish.md` for human
|
|
300
|
+
review β drift is caught at the release, not months later.
|
|
301
|
+
|
|
258
302
|
## Roadmap
|
|
259
303
|
|
|
260
|
-
- **Monthly drift reports** for a curated set of public skills, published under
|
|
261
|
-
`reports/` and at driftproofhq.com.
|
|
262
|
-
- **Model-release triggers** β re-run receipts automatically when a new model
|
|
263
|
-
version ships, so drift is caught at the release, not months later.
|
|
264
304
|
- **A sandboxed-execution harness** for tool-execution skills (document
|
|
265
305
|
renderers, diagram/asset generators) that Report #001 scopes out.
|
|
266
|
-
- **Signed receipts** β key signatures / attestation over the canonical form
|
|
306
|
+
- **Signed receipts** β key signatures / attestation over the canonical form
|
|
307
|
+
(today's `receipt_hash` is tamper-evidence, not authenticity).
|
|
308
|
+
- **More import mappings** β the interop wall (DECLARED vs TESTED) is built;
|
|
309
|
+
adding a converter for another harness's results format is a small,
|
|
310
|
+
well-marked PR (see [docs/interop.md](docs/interop.md)).
|
|
267
311
|
|
|
268
312
|
## Contributing
|
|
269
313
|
|
package/bin/driftproof
CHANGED
|
@@ -9,7 +9,7 @@ const { loadSkill } = require('../lib/skill');
|
|
|
9
9
|
const { runSkillOnModel, summarizeReceipt, projectCalls } = require('../lib/run');
|
|
10
10
|
const { validateReceipt, verifyReceiptHash } = require('../lib/receipt');
|
|
11
11
|
const { buildDriftReport } = require('../lib/diff');
|
|
12
|
-
const {
|
|
12
|
+
const { surfaceForModel, isSubscriptionSurface, resolveModel } = require('../lib/provider');
|
|
13
13
|
const { estimateRunCostUSD, BudgetTracker } = require('../lib/cost');
|
|
14
14
|
const { registryStatus } = require('../lib/models');
|
|
15
15
|
const { verdictFromReceipt, badgeEndpoint, githubOutputLines } = require('../lib/verdict');
|
|
@@ -79,6 +79,8 @@ USAGE
|
|
|
79
79
|
${PROJECT_NAME} diff <receiptA.json> <receiptB.json> [--out FILE]
|
|
80
80
|
${PROJECT_NAME} validate <receipt.json>
|
|
81
81
|
${PROJECT_NAME} badge <receipt.json> [--out FILE] [--github-output]
|
|
82
|
+
${PROJECT_NAME} import <results.json> --from agent-skills-eval|skillgrade [--out DIR]
|
|
83
|
+
${PROJECT_NAME} export <receipt.json> [--to summary-json] [--report-url URL] [--out FILE]
|
|
82
84
|
|
|
83
85
|
ENV
|
|
84
86
|
CLAUDE_PROVIDER api | cli (default: cli β spawns \`claude -p\`, strips ANTHROPIC_API_KEY)
|
|
@@ -104,7 +106,13 @@ NOTES
|
|
|
104
106
|
model still runs but is marked registry:"unregistered" and costed at the
|
|
105
107
|
conservative default price.
|
|
106
108
|
- --concurrency runs that many (case,mode) tasks at once (default 1). Higher
|
|
107
|
-
values cut wall-clock on the cli surface (cold-start dominated)
|
|
109
|
+
values cut wall-clock on the cli surface (cold-start dominated).
|
|
110
|
+
- import converts another tool's results into a valid receipt with HONEST
|
|
111
|
+
epistemics: verification_level DECLARED (never TESTED), surface "external",
|
|
112
|
+
source "imported/<tool>", and NO fabricated hashes. Imported receipts are
|
|
113
|
+
excluded from drift verdicts (diff reports NOT MEASURED). See docs/interop.md.
|
|
114
|
+
- export --to summary-json emits the minimal stable interchange summary
|
|
115
|
+
(driftproof/summary v1) other tools can consume without a receipt parser.`);
|
|
108
116
|
}
|
|
109
117
|
|
|
110
118
|
function slug(s) { return String(s).toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-|-$/g, ''); }
|
|
@@ -137,7 +145,10 @@ async function cmdRun(positional, flags) {
|
|
|
137
145
|
// On `claude-cli` the metered spend is $0 (subscription); the figure is the
|
|
138
146
|
// hypothetical "if run on the metered API" cost β printed, never blocks.
|
|
139
147
|
const cost = estimateRunCostUSD({ caseCount: nCases, samples, models: models.map((m) => require('../lib/provider').resolveModel(m)), judgeModel: judgeModel || 'haiku' });
|
|
140
|
-
|
|
148
|
+
// Surface is per-model now (a run may mix an Anthropic and an OpenAI target).
|
|
149
|
+
const surfaces = [...new Set(models.map((m) => surfaceForModel(m)))];
|
|
150
|
+
const surface = surfaces.join(', ');
|
|
151
|
+
const subSurfaces = surfaces.filter(isSubscriptionSurface);
|
|
141
152
|
|
|
142
153
|
const regStatuses = models.map((m) => `${m}:${registryStatus(m)}`);
|
|
143
154
|
console.log(`\n${PROJECT_NAME} run β skill "${skill.name}" v${skill.version}`);
|
|
@@ -146,7 +157,7 @@ async function cmdRun(positional, flags) {
|
|
|
146
157
|
console.log(` registry: ${regStatuses.join(' ')}${keepTranscripts ? ' transcripts: retained-local' : ''}`);
|
|
147
158
|
console.log(` projected calls: ${perModelCalls}/model Γ ${models.length} model(s) = ${totalProjected} per-model cap: ${maxCalls}`);
|
|
148
159
|
console.log(` projected cost: ~$${cost.totalUSD.toFixed(2)} (rough upper bound; budget $${maxUsd.toFixed(2)}, hard-stop $${(maxUsd * 1.25).toFixed(2)})`);
|
|
149
|
-
if (
|
|
160
|
+
if (subSurfaces.length) console.log(` actual metered spend on ${subSurfaces.join(', ')}: $0.00 (subscription; the $ figure is the estimated-equivalent API cost, counted against the cap identically)`);
|
|
150
161
|
|
|
151
162
|
// Call-count guard: refuse before spending anything if a single model run
|
|
152
163
|
// would exceed the per-run call cap.
|
|
@@ -307,6 +318,55 @@ function cmdBadge(positional, flags) {
|
|
|
307
318
|
}
|
|
308
319
|
}
|
|
309
320
|
|
|
321
|
+
// Convert another tool's results file into a valid DECLARED receipt (interop,
|
|
322
|
+
// Phase 7). The converted receipt is validated + self-hash-verified before it
|
|
323
|
+
// is written; a failed conversion writes nothing.
|
|
324
|
+
function cmdImport(positional, flags) {
|
|
325
|
+
const p = positional[0];
|
|
326
|
+
const from = flags.from;
|
|
327
|
+
const { IMPORT_TOOLS, importResults } = require('../lib/importers');
|
|
328
|
+
if (!p || typeof from !== 'string') {
|
|
329
|
+
console.error(`usage: driftproof import <results.json> --from ${IMPORT_TOOLS.join('|')} [--out DIR]`);
|
|
330
|
+
process.exit(2);
|
|
331
|
+
}
|
|
332
|
+
const data = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
333
|
+
const receipt = importResults(data, { from, importedAt: new Date().toISOString() });
|
|
334
|
+
const { valid, errors } = validateReceipt(receipt);
|
|
335
|
+
if (!valid || !verifyReceiptHash(receipt)) {
|
|
336
|
+
console.error('β converted receipt failed validation β nothing written:', JSON.stringify(errors, null, 2));
|
|
337
|
+
process.exit(1);
|
|
338
|
+
}
|
|
339
|
+
const outDir = flags.out ? path.resolve(flags.out) : RECEIPTS_DIR;
|
|
340
|
+
fs.mkdirSync(outDir, { recursive: true });
|
|
341
|
+
const base = `${slug(receipt.skill.name)}-${slug(receipt.run.model_id)}-imported-${slug(from)}-${dateStamp(receipt.run.date_utc)}`;
|
|
342
|
+
const jsonPath = path.join(outDir, `${base}.json`);
|
|
343
|
+
fs.writeFileSync(jsonPath, JSON.stringify(receipt, null, 2));
|
|
344
|
+
console.log(`imported β ${path.relative(process.cwd(), jsonPath)}`);
|
|
345
|
+
console.log(` verification_level: DECLARED (source ${receipt.run.source}) β the source tool's declaration, converted faithfully;`);
|
|
346
|
+
console.log(` no generation hashes (never fabricated); excluded from drift verdicts unless re-run TESTED. See docs/interop.md.`);
|
|
347
|
+
}
|
|
348
|
+
|
|
349
|
+
// Emit the minimal stable interchange summary for one receipt (interop, Phase 7).
|
|
350
|
+
function cmdExport(positional, flags) {
|
|
351
|
+
const p = positional[0];
|
|
352
|
+
const to = flags.to || 'summary-json';
|
|
353
|
+
if (!p) { console.error('usage: driftproof export <receipt.json> [--to summary-json] [--report-url URL] [--out FILE]'); process.exit(2); }
|
|
354
|
+
if (to !== 'summary-json') { console.error(`unknown export target "${to}" β supported: summary-json`); process.exit(2); }
|
|
355
|
+
const { toSummaryJson } = require('../lib/export');
|
|
356
|
+
const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
357
|
+
if (!verifyReceiptHash(receipt)) console.error(` β ${path.basename(p)}: receipt_hash does not verify (tampered or hand-edited)`);
|
|
358
|
+
const summary = toSummaryJson(receipt, { reportUrl: typeof flags['report-url'] === 'string' ? flags['report-url'] : null });
|
|
359
|
+
const json = JSON.stringify(summary, null, 2);
|
|
360
|
+
if (flags.out) {
|
|
361
|
+
const out = path.resolve(flags.out);
|
|
362
|
+
fs.mkdirSync(path.dirname(out), { recursive: true });
|
|
363
|
+
fs.writeFileSync(out, json + '\n');
|
|
364
|
+
console.log(`summary written to ${flags.out} (${summary.verdict}, delta ${summary.delta === null ? 'n/a' : summary.delta})`);
|
|
365
|
+
} else {
|
|
366
|
+
console.log(json);
|
|
367
|
+
}
|
|
368
|
+
}
|
|
369
|
+
|
|
310
370
|
async function main() {
|
|
311
371
|
const [, , cmd, ...rest] = process.argv;
|
|
312
372
|
const { positional, flags } = parseArgs(rest);
|
|
@@ -316,6 +376,8 @@ async function main() {
|
|
|
316
376
|
case 'diff': return cmdDiff(positional, flags);
|
|
317
377
|
case 'validate': return cmdValidate(positional);
|
|
318
378
|
case 'badge': return cmdBadge(positional, flags);
|
|
379
|
+
case 'import': return cmdImport(positional, flags);
|
|
380
|
+
case 'export': return cmdExport(positional, flags);
|
|
319
381
|
case 'version': case '--version': case '-v': console.log(RUNNER_VERSION); return;
|
|
320
382
|
case 'help': case '--help': case '-h': case undefined: return usage();
|
|
321
383
|
default: console.error(`unknown command: ${cmd}\n`); usage(); process.exit(2);
|
package/config/models.json
CHANGED
|
@@ -1,10 +1,14 @@
|
|
|
1
1
|
{
|
|
2
|
-
"_comment": "Driftproof model registry. The runner resolves --models ids against this list; an unknown id still runs but its receipt is marked registry:\"unregistered\" and its cost is estimated with the conservative default price in lib/models.js. Prices are STANDARD first-party USD per 1,000,000 tokens (input/output) from platform.claude.com (July 2026); Sonnet 5's introductory rate is deliberately NOT used so projections stay an upper bound.
|
|
3
|
-
"registry_version": "1.
|
|
2
|
+
"_comment": "Driftproof model registry. The runner resolves --models ids against this list; an unknown id still runs but its receipt is marked registry:\"unregistered\" and its cost is estimated with the conservative default price in lib/models.js. Prices are STANDARD first-party USD per 1,000,000 tokens (input/output). Anthropic prices are from platform.claude.com (July 2026); Sonnet 5's introductory rate is deliberately NOT used so projections stay an upper bound. OpenAI prices are the public standard per-MTok rates as of July 2026 (see reports/phase-6-providers.md for the fetched source): GPT-5.6 Sol $5/$30, Terra $2.50/$15, Luna $1/$6; GPT-5.5 $5/$30; GPT-5.4 $2.50/$15. gpt-5.4-mini is NOT in the published table (which lists a Nano tier at $0.20/$1.25), so its price here is a deliberately conservative UPPER-bound estimate flagged price_estimate:true. `released` is best-effort (spec RECEIPT.md open question #4). `tier` drives reporting; `judge_eligible` encodes the fixed-judge policy (docs/judge-policy.html) β only the cheap Haiku judge is eligible, and NO OpenAI model is judge-eligible (the judge stays claude-haiku across providers so a cross-substrate report varies only the target model). `provider` is the two-axis provider; per-provider surface/base_url config is under `providers`. The release trigger (scripts/release-watch.js) appends auto-discovered ids here with auto_added:true.",
|
|
3
|
+
"registry_version": "1.1",
|
|
4
4
|
"provider": "anthropic",
|
|
5
|
+
"providers": {
|
|
6
|
+
"anthropic": { "surfaces": ["api", "cli"] },
|
|
7
|
+
"openai": { "base_url": "https://api.openai.com/v1", "api_key_env": "OPENAI_API_KEY", "surfaces": ["api", "cli"] }
|
|
8
|
+
},
|
|
5
9
|
"default_price_note": "Unregistered ids are costed at the most expensive known tier (Fable, $10/$50) β a budget guard must never under-estimate. See lib/models.js DEFAULT_PRICE.",
|
|
6
10
|
"models": [
|
|
7
|
-
{ "id": "claude-fable-5", "family": "fable", "provider": "anthropic", "released":
|
|
11
|
+
{ "id": "claude-fable-5", "family": "fable", "provider": "anthropic", "released": "2026-06-09", "input_price": 10.0, "output_price": 50.0, "tier": "frontier", "judge_eligible": false },
|
|
8
12
|
{ "id": "claude-opus-5", "family": "opus", "provider": "anthropic", "released": null, "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
|
|
9
13
|
{ "id": "claude-opus-4-8", "family": "opus", "provider": "anthropic", "released": null, "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
|
|
10
14
|
{ "id": "claude-opus-4-7", "family": "opus", "provider": "anthropic", "released": null, "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
|
|
@@ -14,6 +18,12 @@
|
|
|
14
18
|
{ "id": "claude-sonnet-4-6", "family": "sonnet", "provider": "anthropic", "released": "2026-02-17", "input_price": 3.0, "output_price": 15.0, "tier": "standard", "judge_eligible": false },
|
|
15
19
|
{ "id": "claude-sonnet-4-5-20250929", "family": "sonnet", "provider": "anthropic", "released": "2025-09-29", "input_price": 3.0, "output_price": 15.0, "tier": "standard", "judge_eligible": false },
|
|
16
20
|
{ "id": "claude-haiku-4-5", "family": "haiku", "provider": "anthropic", "released": "2025-10-01", "input_price": 1.0, "output_price": 5.0, "tier": "cheap", "judge_eligible": true },
|
|
17
|
-
{ "id": "claude-haiku-4-5-20251001", "family": "haiku", "provider": "anthropic", "released": "2025-10-01", "input_price": 1.0, "output_price": 5.0, "tier": "cheap", "judge_eligible": true }
|
|
21
|
+
{ "id": "claude-haiku-4-5-20251001", "family": "haiku", "provider": "anthropic", "released": "2025-10-01", "input_price": 1.0, "output_price": 5.0, "tier": "cheap", "judge_eligible": true },
|
|
22
|
+
{ "id": "gpt-5.6-sol", "family": "gpt-5.6", "provider": "openai", "released": "2026-07-09", "input_price": 5.0, "output_price": 30.0, "tier": "frontier", "judge_eligible": false },
|
|
23
|
+
{ "id": "gpt-5.6-terra", "family": "gpt-5.6", "provider": "openai", "released": "2026-07-09", "input_price": 2.5, "output_price": 15.0, "tier": "standard", "judge_eligible": false },
|
|
24
|
+
{ "id": "gpt-5.6-luna", "family": "gpt-5.6", "provider": "openai", "released": "2026-07-09", "input_price": 1.0, "output_price": 6.0, "tier": "cheap", "judge_eligible": false },
|
|
25
|
+
{ "id": "gpt-5.5", "family": "gpt-5.5", "provider": "openai", "released": null, "input_price": 5.0, "output_price": 30.0, "tier": "frontier", "judge_eligible": false },
|
|
26
|
+
{ "id": "gpt-5.4", "family": "gpt-5.4", "provider": "openai", "released": null, "input_price": 2.5, "output_price": 15.0, "tier": "standard", "judge_eligible": false },
|
|
27
|
+
{ "id": "gpt-5.4-mini", "family": "gpt-5.4", "provider": "openai", "released": null, "input_price": 0.4, "output_price": 2.0, "tier": "cheap", "judge_eligible": false, "price_estimate": true, "price_note": "conservative upper-bound estimate; the public table lists a GPT-5.4 Nano tier ($0.20/$1.25) but not a Mini rate as of July 2026" }
|
|
18
28
|
]
|
|
19
29
|
}
|
package/config.js
CHANGED
|
@@ -9,16 +9,27 @@ const PROJECT_NAME = 'driftproof';
|
|
|
9
9
|
// Bumped whenever the runner's behaviour or receipt-generation semantics change
|
|
10
10
|
// in a way that could affect results. Recorded into every receipt as
|
|
11
11
|
// run.runner_version so a receipt is reproducible against a known engine.
|
|
12
|
-
const RUNNER_VERSION = '0.
|
|
12
|
+
const RUNNER_VERSION = '0.5.0';
|
|
13
13
|
|
|
14
14
|
// The eval format we CONSUME (we deliberately do not invent our own).
|
|
15
15
|
const SUITE_FORMAT = 'agentskills.io/evals';
|
|
16
16
|
|
|
17
17
|
// Receipt schema version this runner emits. Loader/validator accept older
|
|
18
18
|
// versions too (see lib/receipt.js), but new receipts are stamped current.
|
|
19
|
-
// v0.3
|
|
20
|
-
//
|
|
21
|
-
|
|
19
|
+
// v0.3 adds transcript-auditability: per-case generation_hash + judge_sample_
|
|
20
|
+
// hashes[], plus run.registry and run.transcripts.
|
|
21
|
+
// v0.3.1 (additive over v0.3) adds run.provider (two-axis provider), the OpenAI
|
|
22
|
+
// surface enums (openai-api/openai-cli), optional run.surface_overhead_note,
|
|
23
|
+
// optional per-case checks[] (deterministic post-checks), and optional
|
|
24
|
+
// skill.tokens (value-per-token axis). v0.1/v0.2/v0.3 receipts still load.
|
|
25
|
+
// v0.4 (additive over v0.3.1) adds the ECONOMICS axis: per-case, per-arm
|
|
26
|
+
// generation `usage` (input/output/cached tokens + measured wall_ms), a
|
|
27
|
+
// separate per-case `judge_usage` (measurement overhead, excluded from
|
|
28
|
+
// every skill-value figure), run.pricing_snapshot (registry prices frozen
|
|
29
|
+
// at run time so derived dollars stay reproducible), and the derived
|
|
30
|
+
// `economics` block. v0.3.1 is frozen as receipt.v0.3.1.schema.json;
|
|
31
|
+
// v0.1/v0.2/v0.3/v0.3.1 receipts all still load.
|
|
32
|
+
const RECEIPT_SCHEMA_VERSION = '0.4';
|
|
22
33
|
|
|
23
34
|
// Hard USD budget defaults per entry point (Week 4). --max-usd overrides any of
|
|
24
35
|
// these. The projection is refused before any call if it exceeds the cap, on
|
|
@@ -45,4 +56,43 @@ const DEFAULT_JUDGE_SAMPLES = 5;
|
|
|
45
56
|
// spec/RECEIPT.md Β§ "Drift verdict rule" and the report methodology.
|
|
46
57
|
const EFFECT_FLOOR = 0.05;
|
|
47
58
|
|
|
48
|
-
|
|
59
|
+
// Report #002 is the first CROSS-PROVIDER report: the same suites, the same fixed
|
|
60
|
+
// Haiku judge, run on two substrates β a Claude flagship and a GPT flagship. The
|
|
61
|
+
// GPT flagship is a config constant (not hard-coded across scripts) so a future
|
|
62
|
+
// edition swaps one line. Default: the current GPT flagship, gpt-5.6-sol.
|
|
63
|
+
const REPORT_002_CLAUDE_MODEL = 'claude-sonnet-5';
|
|
64
|
+
const REPORT_002_GPT_MODEL = 'gpt-5.6-sol';
|
|
65
|
+
const REPORT_002_JUDGE_MODEL = 'claude-haiku-4-5';
|
|
66
|
+
|
|
67
|
+
// Report #003 is a RELEASE DRIFT report (Report #001's type): one provider, two
|
|
68
|
+
// model versions, verdicts REGRESSED/IMPROVED/MIXED/WITHIN NOISE per skill under
|
|
69
|
+
// the per-case band-separation rule + effect floor. New vs its family predecessor.
|
|
70
|
+
const REPORT_003_NEW_MODEL = 'claude-opus-5';
|
|
71
|
+
const REPORT_003_OLD_MODEL = 'claude-opus-4-8';
|
|
72
|
+
const REPORT_003_JUDGE_MODEL = 'claude-haiku-4-5';
|
|
73
|
+
|
|
74
|
+
// Report #004 is a CAPABILITY-GAP report (Report #002's verdict style, ONE
|
|
75
|
+
// provider, TWO tiers): fable-5 has NO family predecessor, so this is a
|
|
76
|
+
// cross-family study, NOT release drift. Per skill: the with/without delta on
|
|
77
|
+
// EACH model with bands; tier comparison is context. Headline question: does
|
|
78
|
+
// encoded expertise still lift output on the frontier tier?
|
|
79
|
+
const REPORT_004_BASE_MODEL = 'claude-opus-5'; // flagship tier
|
|
80
|
+
const REPORT_004_FRONTIER_MODEL = 'claude-fable-5'; // frontier tier (full id β no alias)
|
|
81
|
+
const REPORT_004_JUDGE_MODEL = 'claude-haiku-4-5';
|
|
82
|
+
|
|
83
|
+
// Report #005 is a VALUE report β the fourth report type. It asks what a skill
|
|
84
|
+
// COSTS to run alongside whether it helps, over three substrates, and shows the
|
|
85
|
+
// three axes (accuracy lift / Ξcost / Ξlatency) side by side and never combined.
|
|
86
|
+
// Same suites, same fixed judge; the substrate list spans two providers so the
|
|
87
|
+
// economics are read across surfaces, not within one vendor's pricing.
|
|
88
|
+
const REPORT_005_MODELS = ['claude-sonnet-5', 'claude-fable-5', 'gpt-5.6-sol'];
|
|
89
|
+
const REPORT_005_JUDGE_MODEL = 'claude-haiku-4-5';
|
|
90
|
+
|
|
91
|
+
module.exports = {
|
|
92
|
+
PROJECT_NAME, RUNNER_VERSION, SUITE_FORMAT, RECEIPT_SCHEMA_VERSION, DEFAULT_JUDGE_SAMPLES,
|
|
93
|
+
EFFECT_FLOOR, DEV_MAX_USD, REPORT_MAX_USD, TRIGGER_MAX_USD,
|
|
94
|
+
REPORT_002_CLAUDE_MODEL, REPORT_002_GPT_MODEL, REPORT_002_JUDGE_MODEL,
|
|
95
|
+
REPORT_003_NEW_MODEL, REPORT_003_OLD_MODEL, REPORT_003_JUDGE_MODEL,
|
|
96
|
+
REPORT_004_BASE_MODEL, REPORT_004_FRONTIER_MODEL, REPORT_004_JUDGE_MODEL,
|
|
97
|
+
REPORT_005_MODELS, REPORT_005_JUDGE_MODEL,
|
|
98
|
+
};
|
package/lib/checks.js
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// Deterministic post-checks.
|
|
5
|
+
//
|
|
6
|
+
// A per-case eval suite may declare optional `checks[]`: structural / regex
|
|
7
|
+
// assertions on the model OUTPUT that either hold or don't β no LLM judgment.
|
|
8
|
+
// They run ALONGSIDE the sampled judge and are reported as a SEPARATE column.
|
|
9
|
+
//
|
|
10
|
+
// IMPORTANT (scope): post-checks are SUPPLEMENTARY EVIDENCE ONLY. They are NOT
|
|
11
|
+
// folded into the case `outcome` or the band-overlap drift verdict β a check is a
|
|
12
|
+
// cheap, unambiguous signal ("did the output contain a Conventional-Commits
|
|
13
|
+
// subject?") that corroborates or contradicts the judge, not a second grader. A
|
|
14
|
+
// suite author adds them where they are natural and in-text-groundable; they are
|
|
15
|
+
// never required.
|
|
16
|
+
//
|
|
17
|
+
// Supported kinds (kept small and unambiguous):
|
|
18
|
+
// regex β the pattern (with optional `flags`) matches the output
|
|
19
|
+
// contains β the output includes the literal `value` substring
|
|
20
|
+
// not_contains β the output does NOT include the literal `value` substring
|
|
21
|
+
// min_length β the trimmed output is at least `value` characters long
|
|
22
|
+
|
|
23
|
+
function runOneCheck(check, output) {
|
|
24
|
+
const text = String(output || '');
|
|
25
|
+
switch (check && check.kind) {
|
|
26
|
+
case 'regex': {
|
|
27
|
+
let re;
|
|
28
|
+
try { re = new RegExp(check.pattern, check.flags || ''); } catch (_e) { return false; }
|
|
29
|
+
return re.test(text);
|
|
30
|
+
}
|
|
31
|
+
case 'contains': return text.includes(String(check.value));
|
|
32
|
+
case 'not_contains': return !text.includes(String(check.value));
|
|
33
|
+
case 'min_length': return text.trim().length >= Number(check.value || 0);
|
|
34
|
+
default: return false;
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
// Run every declared check against one output. Returns a compact result array
|
|
39
|
+
// [{ name, kind, pass }] suitable for the receipt (v0.3.1 optional per-case
|
|
40
|
+
// `checks`). Empty array when the case declares no checks.
|
|
41
|
+
function runChecks(output, checks) {
|
|
42
|
+
if (!Array.isArray(checks) || !checks.length) return [];
|
|
43
|
+
return checks.map((c) => ({
|
|
44
|
+
name: String((c && c.name) || (c && c.kind) || 'check'),
|
|
45
|
+
kind: c && c.kind,
|
|
46
|
+
pass: !!runOneCheck(c, output),
|
|
47
|
+
}));
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
module.exports = { runChecks, runOneCheck };
|
package/lib/diff.js
CHANGED
|
@@ -73,15 +73,24 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B' } = {}) {
|
|
|
73
73
|
const bB = withSkillBands(b);
|
|
74
74
|
const ids = [...new Set([...Object.keys(aB), ...Object.keys(bB)])];
|
|
75
75
|
|
|
76
|
+
// Interop (Phase 7): drift verdicts require TESTED receipts on BOTH sides.
|
|
77
|
+
// An imported (DECLARED) receipt is a faithful record of another tool's
|
|
78
|
+
// declaration β its numbers are shown, but band-verified regression/
|
|
79
|
+
// improvement verdicts are never computed from evidence we did not run.
|
|
80
|
+
const levelOf = (r) => r.verification_level || 'TESTED';
|
|
81
|
+
const belowTested = [[labelA, a], [labelB, b]].filter(([, r]) => levelOf(r) !== 'TESTED');
|
|
82
|
+
const measured = belowTested.length === 0;
|
|
83
|
+
|
|
76
84
|
const perCase = ids.map((id) => {
|
|
77
85
|
const before = aB[id] || null;
|
|
78
86
|
const after = bB[id] || null;
|
|
79
87
|
const delta = (before && after) ? round(after.mean - before.mean) : null;
|
|
80
|
-
const verdict =
|
|
88
|
+
const verdict = !measured ? 'not measured'
|
|
89
|
+
: (before && after) ? verdictWithFloor(before, after, delta) : 'n/a';
|
|
81
90
|
return { id, before, after, delta, verdict };
|
|
82
91
|
});
|
|
83
92
|
// Sort worst-first: regressions, then by delta.
|
|
84
|
-
const order = { regression: 0, 'within noise': 1, [WITHIN_NOISE_FLOOR]: 1, improvement: 2, 'n/a': 3 };
|
|
93
|
+
const order = { regression: 0, 'within noise': 1, [WITHIN_NOISE_FLOOR]: 1, improvement: 2, 'n/a': 3, 'not measured': 3 };
|
|
85
94
|
perCase.sort((x, y) => (order[x.verdict] - order[y.verdict]) || ((x.delta || 0) - (y.delta || 0)));
|
|
86
95
|
|
|
87
96
|
const aAgg = aggWithBand(a);
|
|
@@ -112,6 +121,13 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B' } = {}) {
|
|
|
112
121
|
if (a.suite.suite_hash !== b.suite.suite_hash) warnings.push('suite_hash differs β the eval suite changed; per-case comparison may be misleading.');
|
|
113
122
|
if (a.skill.name !== b.skill.name) warnings.push(`different skills (${a.skill.name} vs ${b.skill.name}) β comparison is not meaningful.`);
|
|
114
123
|
if ((a.run.judge || {}).samples <= 1 || (b.run.judge || {}).samples <= 1) warnings.push('one or both receipts are single-sample (no bands) β non-overlap can only be trusted when both sides are sampled.');
|
|
124
|
+
if (!measured) warnings.push(`verdicts NOT computed β ${belowTested.map(([l, r]) => `${l} is ${levelOf(r)}${r.run && r.run.source ? ` (${r.run.source})` : ''}`).join('; ')}. Drift verdicts require TESTED receipts on both sides; declared numbers are shown as context only (see /interop.html).`);
|
|
125
|
+
// Cross-provider / cross-surface disclosure (Phase 6). A comparison across
|
|
126
|
+
// providers is a skill-DURABILITY comparison across substrates, not model drift
|
|
127
|
+
// over time; across surfaces, sampling control differs. Both are flagged so a
|
|
128
|
+
// reader never mistakes one for the other (see docs/neutrality.html).
|
|
129
|
+
if ((a.run.provider || 'anthropic') !== (b.run.provider || 'anthropic')) warnings.push(`different providers (${a.run.provider || 'anthropic'} vs ${b.run.provider || 'anthropic'}) β this is a cross-substrate durability comparison, not model drift over time; read the delta, not absolute scores (see the neutrality policy).`);
|
|
130
|
+
if (a.run.surface !== b.run.surface) warnings.push(`different surfaces (${a.run.surface} vs ${b.run.surface}) β sampling control differs between surfaces; compare with care.`);
|
|
115
131
|
if (warnings.length) {
|
|
116
132
|
L.push('> **β Caveats**');
|
|
117
133
|
for (const w of warnings) L.push(`> - ${w}`);
|
|
@@ -122,17 +138,22 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B' } = {}) {
|
|
|
122
138
|
const nFloor = perCase.filter((r) => r.verdict === WITHIN_NOISE_FLOOR).length;
|
|
123
139
|
L.push(`## Headline`);
|
|
124
140
|
L.push('');
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
141
|
+
if (!measured) {
|
|
142
|
+
L.push(`**NOT MEASURED β ${belowTested.length} receipt(s) below TESTED. Drift verdicts require TESTED receipts on both sides; the declared numbers above are context, not band-verified evidence.**`);
|
|
143
|
+
L.push('');
|
|
144
|
+
} else {
|
|
145
|
+
L.push(`**${headlineVerdict(perCase)}**`);
|
|
146
|
+
L.push('');
|
|
147
|
+
L.push(`with_skill mean moved ${fmt(headlineDelta)} (${bandStr(aAgg)} β ${bandStr(bAgg)}; band = suite dispersion). Per-case band-overlap verdicts: ${regressions.length} regression(s), ${perCase.filter((r) => r.verdict === 'improvement').length} improvement(s), ${nWithin} within noise${nFloor ? ` (${nFloor} of them band-separated but below the ${EFFECT_FLOOR} effect floor)` : ''}.`);
|
|
148
|
+
L.push('');
|
|
149
|
+
}
|
|
129
150
|
|
|
130
151
|
L.push(`## Per-case with_skill (band overlap β verdict)`);
|
|
131
152
|
L.push('');
|
|
132
153
|
L.push(`| case | ${labelA} (mean Β± sd) | ${labelB} (mean Β± sd) | Ξ | verdict |`);
|
|
133
154
|
L.push(`|---|---|---|---|---|`);
|
|
134
155
|
for (const r of perCase) {
|
|
135
|
-
const flag = r.verdict === 'regression' ? 'π» regression' : r.verdict === 'improvement' ? 'πΌ improvement' : r.verdict === WITHIN_NOISE_FLOOR ? 'within noise (below floor)' : r.verdict === 'within noise' ? 'within noise' : 'n/a';
|
|
156
|
+
const flag = r.verdict === 'regression' ? 'π» regression' : r.verdict === 'improvement' ? 'πΌ improvement' : r.verdict === WITHIN_NOISE_FLOOR ? 'within noise (below floor)' : r.verdict === 'within noise' ? 'within noise' : r.verdict === 'not measured' ? 'not measured' : 'n/a';
|
|
136
157
|
L.push(`| \`${r.id}\` | ${r.before ? bandStr(r.before) : 'n/a'} | ${r.after ? bandStr(r.after) : 'n/a'} | ${fmt(r.delta)} | ${flag} |`);
|
|
137
158
|
}
|
|
138
159
|
L.push('');
|
package/lib/export.js
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// Receipt interop β export (Phase 7). The lightweight interchange: a minimal,
|
|
5
|
+
// STABLE, flat JSON summary of one receipt (skill, model, bands, delta,
|
|
6
|
+
// verdict) that other tools can consume without a receipt parser. The
|
|
7
|
+
// "driftproof/summary" v1 keys are frozen; additions bump format_version.
|
|
8
|
+
// Documented in docs/interop.md; snapshot-tested in the gate.
|
|
9
|
+
|
|
10
|
+
const { verdictFromReceipt } = require('./verdict');
|
|
11
|
+
|
|
12
|
+
const SUMMARY_FORMAT = 'driftproof/summary';
|
|
13
|
+
const SUMMARY_FORMAT_VERSION = '1';
|
|
14
|
+
|
|
15
|
+
// Build the summary object for one receipt. Deterministic for a given receipt:
|
|
16
|
+
// fixed key order, no export-time timestamps. `reportUrl` is caller-supplied
|
|
17
|
+
// (receipts do not know where their report lives), else null.
|
|
18
|
+
function toSummaryJson(receipt, { reportUrl = null } = {}) {
|
|
19
|
+
const agg = receipt.results.aggregates;
|
|
20
|
+
const cmp = receipt.comparison || {};
|
|
21
|
+
const hasBaseline = agg.baseline && agg.baseline.case_count > 0;
|
|
22
|
+
const v = verdictFromReceipt(receipt);
|
|
23
|
+
return {
|
|
24
|
+
format: SUMMARY_FORMAT,
|
|
25
|
+
format_version: SUMMARY_FORMAT_VERSION,
|
|
26
|
+
skill: { name: receipt.skill.name, version: receipt.skill.version },
|
|
27
|
+
model: {
|
|
28
|
+
id: receipt.run.model_id,
|
|
29
|
+
provider: receipt.run.provider || 'anthropic',
|
|
30
|
+
surface: receipt.run.surface,
|
|
31
|
+
},
|
|
32
|
+
run_date_utc: receipt.run.date_utc,
|
|
33
|
+
scores: {
|
|
34
|
+
with_skill: { mean: agg.with_skill.mean_score, stddev: agg.with_skill.stddev },
|
|
35
|
+
baseline: hasBaseline ? { mean: agg.baseline.mean_score, stddev: agg.baseline.stddev } : null,
|
|
36
|
+
},
|
|
37
|
+
delta: typeof cmp.delta === 'number' ? cmp.delta : null,
|
|
38
|
+
delta_uncertainty: typeof cmp.delta_uncertainty === 'number' ? cmp.delta_uncertainty : null,
|
|
39
|
+
verdict: v.verdict,
|
|
40
|
+
verification_level: receipt.verification_level,
|
|
41
|
+
source: receipt.run.source || 'driftproof',
|
|
42
|
+
judge: {
|
|
43
|
+
model_id: (receipt.results.cases.find((c) => c.judge) || { judge: { model_id: 'unknown' } }).judge.model_id,
|
|
44
|
+
samples: (receipt.run.judge || {}).samples || 1,
|
|
45
|
+
},
|
|
46
|
+
receipt_hash: receipt.receipt_hash,
|
|
47
|
+
report_url: reportUrl,
|
|
48
|
+
spec: 'https://driftproofhq.com/spec/receipt.schema.json',
|
|
49
|
+
};
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
module.exports = { toSummaryJson, SUMMARY_FORMAT, SUMMARY_FORMAT_VERSION };
|