driftproof 0.11.2 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +79 -12
- package/bin/driftproof +343 -39
- package/config/models.json +3 -2
- package/config.js +2 -2
- package/lib/counts.js +104 -0
- package/lib/decision.js +72 -31
- package/lib/diff.js +12 -3
- package/lib/export.js +6 -2
- package/lib/importers-anthropic.js +451 -0
- package/lib/importers.js +50 -13
- package/lib/init.js +3 -2
- package/lib/receipt.js +182 -10
- package/lib/regrade.js +266 -0
- package/lib/reuse.js +144 -9
- package/lib/run.js +39 -4
- package/lib/skill.js +70 -15
- package/lib/stale.js +194 -0
- package/lib/verdict.js +50 -16
- package/package.json +1 -1
- package/spec/RECEIPT.md +103 -13
- package/spec/receipt.schema.json +330 -15
- package/spec/receipt.v0.7.schema.json +1563 -0
- package/spec/receipt.v0.8.schema.json +1677 -0
- package/spec/stale.v1.schema.json +353 -0
package/bin/driftproof
CHANGED
|
@@ -8,7 +8,7 @@ const { PROJECT_NAME, RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES, DEV_MAX_USD, DEV_MA
|
|
|
8
8
|
const { loadSkill } = require('../lib/skill');
|
|
9
9
|
const { runSkillOnModel, summarizeReceipt, projectCalls, answeredLine, band, uncertaintyStr } = require('../lib/run');
|
|
10
10
|
const { SAMPLING } = require('../lib/sampling');
|
|
11
|
-
const { validateReceipt, verifyReceiptHash } = require('../lib/receipt');
|
|
11
|
+
const { validateReceipt, verifyReceiptHash, duplicateCaseRows, ambiguityLine, listReceipts, receiptBaseName, fileSlug, REGRADE_SIDECAR, SUMMARY_SIDECAR } = require('../lib/receipt');
|
|
12
12
|
const { buildDriftReport, revisionPairProblem } = require('../lib/diff');
|
|
13
13
|
const { surfaceForModel, isSubscriptionSurface, resolveModel, evalUser } = require('../lib/provider');
|
|
14
14
|
const { estimateRunCostUSD, BudgetTracker } = require('../lib/cost');
|
|
@@ -16,6 +16,8 @@ const { registryStatus, assertRegistered, REGISTRY_PATH } = require('../lib/mode
|
|
|
16
16
|
const { verdictFromReceipt, badgeEndpoint, githubOutputLines, drawsLine, UNDERPOWERED_LINE } = require('../lib/verdict');
|
|
17
17
|
const decision = require('../lib/decision');
|
|
18
18
|
const { scaffoldInit } = require('../lib/init');
|
|
19
|
+
const { planRegrade, regradeReceipt } = require('../lib/regrade');
|
|
20
|
+
const { sha256 } = require('../lib/canonical');
|
|
19
21
|
|
|
20
22
|
// Output dirs default to the USER's current directory, not the package dir, so a
|
|
21
23
|
// global/npx install writes receipts/transcripts into the project being tested
|
|
@@ -31,7 +33,7 @@ function writeTranscripts(receipt, transcripts) {
|
|
|
31
33
|
const manifest = { receipt_hash: receipt.receipt_hash, model_id: receipt.run.model_id, date_utc: receipt.run.date_utc, entries: [] };
|
|
32
34
|
for (const t of transcripts) {
|
|
33
35
|
if (!t) continue;
|
|
34
|
-
const base = `${
|
|
36
|
+
const base = `${fileSlug(t.id)}-${t.mode}`;
|
|
35
37
|
fs.writeFileSync(path.join(dir, `${base}.json`), JSON.stringify({ id: t.id, mode: t.mode, generation: t.generation, judge_outputs: t.judge_outputs }, null, 2));
|
|
36
38
|
manifest.entries.push(`${base}.json`);
|
|
37
39
|
}
|
|
@@ -59,24 +61,81 @@ function refuseUnverified(receipt, file, command) {
|
|
|
59
61
|
process.exit(4);
|
|
60
62
|
}
|
|
61
63
|
|
|
62
|
-
//
|
|
63
|
-
//
|
|
64
|
-
//
|
|
64
|
+
// ── AN AMBIGUOUS RECEIPT IS NOT RENDERED (spec 050 AC-3) ─────────────────────
|
|
65
|
+
// A receipt with two rows for one (id, mode) verifies and, before spec 050, validated:
|
|
66
|
+
// the loader let two cases share an id, and the verdict kept whichever row came last.
|
|
67
|
+
// The surfaces that render ONE receipt refuse it here, after the hash check, the way
|
|
68
|
+
// they refuse a tampered one: nothing on stdout, no file, the pair on stderr, exit 2.
|
|
69
|
+
// The set surfaces (decide, badge <dir>) do not come here: they decide the model
|
|
70
|
+
// `refused`, which fails the job closed and keeps the row that says why.
|
|
71
|
+
function refuseAmbiguous(receipt, file, command) {
|
|
72
|
+
const dups = duplicateCaseRows(receipt);
|
|
73
|
+
if (!dups.length) return;
|
|
74
|
+
console.error(` ✗ REFUSED (${command}): ${path.basename(file)} is ambiguous: ${ambiguityLine(dups)}. A verdict over it would depend on which row was read; nothing rendered.`);
|
|
75
|
+
process.exit(2);
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
// ── A RECEIPT THAT DOES NOT VALIDATE IS NOT RENDERED (spec 062, register row 2) ─
|
|
79
|
+
// badge and decide verified the hash and never validated, so a sealed receipt whose
|
|
80
|
+
// schema_version named an Object.prototype property rendered `passing` and decided
|
|
81
|
+
// `helped`. Called after refuseUnverified, and after refuseAmbiguous where a surface
|
|
82
|
+
// calls that, so spec 050's exit 2 for an ambiguous receipt stands: nothing on
|
|
83
|
+
// stdout, no file, the reason on stderr naming the file, exit 4 as for a receipt
|
|
84
|
+
// whose hash does not verify.
|
|
85
|
+
function refuseInvalid(receipt, file, command) {
|
|
86
|
+
const { valid, errors } = validateReceipt(receipt);
|
|
87
|
+
if (valid) return;
|
|
88
|
+
const first = errors[0] || {};
|
|
89
|
+
console.error(` ✗ REFUSED (${command}): ${path.basename(file)} does not validate against its schema (${errors.length} error(s); ${first.instancePath || '/'} ${first.message || ''}); nothing rendered. Run driftproof validate on it for the list.`);
|
|
90
|
+
process.exit(4);
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
// Read an optional .driftproofrc (JSON) for per-project run defaults: the CWD's,
|
|
94
|
+
// then the skill dir's (where `driftproof init` writes it). CLI flags always win
|
|
95
|
+
// over the rc; the rc wins over built-in defaults.
|
|
96
|
+
//
|
|
97
|
+
// Spec 062 (register row 6): an rc that does not parse is REFUSED, exit 2, naming
|
|
98
|
+
// the file. It used to be ignored, so a trailing comma ran the project with the
|
|
99
|
+
// built-in budget and caps instead of the ones the file set.
|
|
100
|
+
//
|
|
101
|
+
// Spec 062 (register row 4): the skill directory is the skill's content (in the
|
|
102
|
+
// Action, pull-request content), and it may not narrow the run that measures it.
|
|
103
|
+
// max_cases, samples and judge_model are read from the working directory's rc
|
|
104
|
+
// only; set in the skill directory's they are ignored and named on stderr. When
|
|
105
|
+
// the working directory IS the skill directory, its rc is the skill's.
|
|
106
|
+
//
|
|
107
|
+
// Spec 069: in the Action the checkout's root is pull-request content too, so
|
|
108
|
+
// action/run.sh starts `run` from an empty directory and no working-directory
|
|
109
|
+
// rc is read there at all.
|
|
110
|
+
const SKILL_RC_IGNORED = ['max_cases', 'samples', 'judge_model'];
|
|
111
|
+
function readRc(p) {
|
|
112
|
+
if (!fs.existsSync(p)) return {};
|
|
113
|
+
let rc;
|
|
114
|
+
try { rc = JSON.parse(fs.readFileSync(p, 'utf8')); } catch (e) {
|
|
115
|
+
console.error(` ✗ REFUSED: ${p} is not valid JSON (${e.message}); nothing run. Fix the file or remove it.`);
|
|
116
|
+
process.exit(2);
|
|
117
|
+
}
|
|
118
|
+
if (!rc || typeof rc !== 'object' || Array.isArray(rc)) {
|
|
119
|
+
console.error(` ✗ REFUSED: ${p} is not a JSON object; nothing run. Fix the file or remove it.`);
|
|
120
|
+
process.exit(2);
|
|
121
|
+
}
|
|
122
|
+
return rc;
|
|
123
|
+
}
|
|
65
124
|
function loadRc(skillDir) {
|
|
66
125
|
const merged = {};
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
126
|
+
const skillRcDir = skillDir ? path.resolve(skillDir) : null;
|
|
127
|
+
if (path.resolve(process.cwd()) !== skillRcDir) Object.assign(merged, readRc(path.join(process.cwd(), '.driftproofrc')));
|
|
128
|
+
if (skillRcDir) {
|
|
129
|
+
const p = path.join(skillRcDir, '.driftproofrc');
|
|
130
|
+
const rc = readRc(p);
|
|
131
|
+
const dropped = SKILL_RC_IGNORED.filter((k) => Object.hasOwn(rc, k));
|
|
132
|
+
for (const k of dropped) delete rc[k];
|
|
133
|
+
if (dropped.length) console.error(` note: ${p} sets ${dropped.join(', ')}; ignored, because a skill directory may not narrow the run that measures it (set them as flags, or in the working directory's .driftproofrc when you run the CLI yourself; the GitHub Action reads no working-directory .driftproofrc)`);
|
|
134
|
+
Object.assign(merged, rc);
|
|
72
135
|
}
|
|
73
136
|
return merged;
|
|
74
137
|
}
|
|
75
138
|
|
|
76
|
-
// Flags that never take a value, so `--trusted-skill <skill-dir>` keeps the dir
|
|
77
|
-
// positional instead of swallowing it as the flag's value.
|
|
78
|
-
const BOOLEAN_FLAGS = new Set(['trusted-skill', 'svg']);
|
|
79
|
-
|
|
80
139
|
// ── THE INPUT CONTRACT (spec 026 AC-9, F5) ────────────────────────────────────
|
|
81
140
|
// A COPY of action/lib.sh's RE_* rules, in the shape spec 028's fixture holds
|
|
82
141
|
// them; specs/026-receipt-integrity/fixtures/input-contract.json carries the
|
|
@@ -152,21 +211,75 @@ function checkNumericInputs(flags, rc) {
|
|
|
152
211
|
return out;
|
|
153
212
|
}
|
|
154
213
|
|
|
214
|
+
// Spec 062 (register row 6): `--name=value` is the flag `name` with the value
|
|
215
|
+
// `value`, the empty value included. It used to be a flag named `name=value`, which
|
|
216
|
+
// no command read, so `--max-usd=0.01` ran with the default budget.
|
|
217
|
+
//
|
|
218
|
+
// A flag that takes no value is set by its name alone, so `--trusted-skill <skill-dir>`
|
|
219
|
+
// keeps the dir positional instead of swallowing it as the flag's value (spec 106: all
|
|
220
|
+
// eight, where three used to take the next argument). Given with `=`, or followed by a
|
|
221
|
+
// boolean word, it is refused, exit 2, naming it, for every command (A-062-3, spec 106):
|
|
222
|
+
// `--trusted-skill=false` and `--trusted-skill false` each turned the same-user lane on.
|
|
223
|
+
const NO_VALUE_FLAGS = new Set(['trusted-skill', 'svg', 'count-errored-runs', 'strict', 'no-harness-check', 'keep-transcripts', 'github-output', 'enforce']);
|
|
224
|
+
const BOOLEAN_WORD = /^(true|false|yes|no|on|off|1|0)$/i;
|
|
225
|
+
function refuseValue(key, given) {
|
|
226
|
+
console.error(` \u2717 REFUSED: --${key} takes no value, and was given one (${given}). Give --${key} alone to set it, or leave it out. Nothing run, no receipt written.`);
|
|
227
|
+
process.exit(2);
|
|
228
|
+
}
|
|
155
229
|
function parseArgs(argv) {
|
|
156
230
|
const positional = [];
|
|
157
231
|
const flags = {};
|
|
158
232
|
for (let i = 0; i < argv.length; i++) {
|
|
159
233
|
const a = argv[i];
|
|
234
|
+
const eq = a.startsWith('--') ? a.indexOf('=') : -1;
|
|
235
|
+
if (eq > 2 && NO_VALUE_FLAGS.has(a.slice(2, eq))) refuseValue(a.slice(2, eq), a);
|
|
236
|
+
if (eq > 2) { flags[a.slice(2, eq)] = a.slice(eq + 1); continue; }
|
|
160
237
|
if (a.startsWith('--')) {
|
|
161
238
|
const key = a.slice(2);
|
|
162
239
|
const next = argv[i + 1];
|
|
163
|
-
if (
|
|
240
|
+
if (NO_VALUE_FLAGS.has(key) && next !== undefined && BOOLEAN_WORD.test(next)) refuseValue(key, `${a} ${next}`);
|
|
241
|
+
if (NO_VALUE_FLAGS.has(key) || next === undefined || next.startsWith('--')) { flags[key] = true; }
|
|
164
242
|
else { flags[key] = next; i++; }
|
|
165
243
|
} else positional.push(a);
|
|
166
244
|
}
|
|
167
245
|
return { positional, flags };
|
|
168
246
|
}
|
|
169
247
|
|
|
248
|
+
// Spec 062 (register row 6): the two commands that spend refuse a flag they do not
|
|
249
|
+
// take, exit 2, naming it, before anything is read or printed. A misspelt cap flag
|
|
250
|
+
// used to be read by nothing, and the run went ahead with the default cap. The
|
|
251
|
+
// `node:util` parseArgs rewrite, one table per command, is later work.
|
|
252
|
+
const COMMAND_FLAGS = {
|
|
253
|
+
run: ['models', 'samples', 'max-cases', 'max-calls', 'judge-model', 'concurrency', 'max-usd', 'keep-transcripts', 'out', 'trusted-skill'],
|
|
254
|
+
regrade: ['skill', 'answers', 'judge-model', 'samples', 'max-calls', 'max-usd', 'out', 'trusted-skill'],
|
|
255
|
+
};
|
|
256
|
+
function refuseUnknownFlags(command, flags) {
|
|
257
|
+
const unknown = Object.keys(flags).filter((k) => !COMMAND_FLAGS[command].includes(k));
|
|
258
|
+
if (!unknown.length) return;
|
|
259
|
+
console.error(` ✗ REFUSED (${command}): unknown flag ${unknown.map((k) => `--${k}`).join(', ')}; ${command} takes ${COMMAND_FLAGS[command].map((k) => `--${k}`).join(', ')}. Nothing run, no receipt written.`);
|
|
260
|
+
process.exit(2);
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
// Spec 106: the same two commands take one argument. Another positional is refused, exit 2,
|
|
264
|
+
// naming each one given and the one taken, before anything is read. A word after a flag that
|
|
265
|
+
// takes no value (`--trusted-skill maybe`), or a single-dash `-trusted-skill`, used to be
|
|
266
|
+
// dropped with no word.
|
|
267
|
+
function refuseExtraPositionals(command, positional, placeholder) {
|
|
268
|
+
if (positional.length <= 1) return;
|
|
269
|
+
console.error(` ✗ REFUSED (${command}): ${command} takes one argument and was given ${positional.length}: ${positional.join(', ')}. It took ${positional[0]} as ${placeholder}, and would read none of ${positional.slice(1).join(', ')}. A flag that takes no value is given alone. Nothing run, no receipt written.`);
|
|
270
|
+
process.exit(2);
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
// A skill that cannot be loaded because a link stands where it is read (spec 062,
|
|
274
|
+
// register row 5) is refused like any other input, naming the path.
|
|
275
|
+
function loadSkillOrRefuse(skillDir, command) {
|
|
276
|
+
try { return loadSkill(skillDir); } catch (e) {
|
|
277
|
+
if (!e || e.code !== 'SKILL_LINK') throw e;
|
|
278
|
+
console.error(` ✗ REFUSED (${command}): ${e.message}. Nothing measured, no receipt written.`);
|
|
279
|
+
process.exit(2);
|
|
280
|
+
}
|
|
281
|
+
}
|
|
282
|
+
|
|
170
283
|
function usage() {
|
|
171
284
|
console.log(`${PROJECT_NAME} v${RUNNER_VERSION} — continuous verification of agent skills
|
|
172
285
|
|
|
@@ -176,12 +289,17 @@ USAGE
|
|
|
176
289
|
[--judge-model M] [--concurrency N] [--max-usd N]
|
|
177
290
|
[--keep-transcripts] [--out DIR] [--trusted-skill]
|
|
178
291
|
${PROJECT_NAME} diff <receiptA.json> <receiptB.json> [--out FILE] [--mode release|revision]
|
|
292
|
+
${PROJECT_NAME} regrade <receipt.json> --skill <dir> --answers <file> --judge-model <id>
|
|
293
|
+
[--samples N] [--max-calls N] [--max-usd N] [--out DIR] [--trusted-skill]
|
|
294
|
+
${PROJECT_NAME} stale <receipt.json>... [--skill DIR] [--suite FILE] [--model ID] [--judge ID]
|
|
295
|
+
[--harness-version V | --no-harness-check] [--strict] [--json [path]]
|
|
179
296
|
${PROJECT_NAME} validate <receipt.json>
|
|
180
297
|
${PROJECT_NAME} badge <receipt.json|receipts-dir> [--models a,b] [--out FILE] [--github-output] [--svg [--href URL]]
|
|
181
298
|
${PROJECT_NAME} decide <receipts-dir> --models a,b [--github-output] [--badge FILE]
|
|
182
299
|
[--summary FILE] [--enforce] [--fail-on-regression true|false]
|
|
183
300
|
${PROJECT_NAME} import <results.json> --from agent-skills-eval|skillgrade [--out DIR]
|
|
184
|
-
${PROJECT_NAME}
|
|
301
|
+
${PROJECT_NAME} import <file|dir> --from claude-plugin-eval|skill-creator [--count-errored-runs] [--out DIR]
|
|
302
|
+
${PROJECT_NAME} export <receipt.json> [--to summary-json] [--report-url URL] [--out FILE|DIR]
|
|
185
303
|
|
|
186
304
|
ENV
|
|
187
305
|
CLAUDE_PROVIDER api | cli (default: cli — spawns \`claude -p\`, strips ANTHROPIC_API_KEY)
|
|
@@ -228,16 +346,38 @@ NOTES
|
|
|
228
346
|
epistemics: verification_level DECLARED (never TESTED), surface "external",
|
|
229
347
|
source "imported/<tool>", and NO fabricated hashes. Imported receipts are
|
|
230
348
|
excluded from drift verdicts (diff reports NOT MEASURED). See docs/interop.md.
|
|
349
|
+
- import --from claude-plugin-eval reads \`claude plugin eval\`'s aggregate-result.json
|
|
350
|
+
(or its --json output); --from skill-creator reads a benchmark.json (skill-creator's,
|
|
351
|
+
or skill-up's, with its result.json beside it). Given a directory, it finds the one
|
|
352
|
+
document and refuses none or several. Runs the source recorded an error for are kept
|
|
353
|
+
out of the band and listed (--count-errored-runs puts them back); anything the source
|
|
354
|
+
does not record is written as unknown and named in run.import.notices.
|
|
355
|
+
- stale compares what each receipt recorded (model, harness, skill, suite, judge,
|
|
356
|
+
template, each case's rubric) with what would run today, read from the flags,
|
|
357
|
+
then .driftproofrc; a value nobody gives is unknown, never current. It says per
|
|
358
|
+
arm whether the result stands, needs a rerun, or only a regrade, and prints the
|
|
359
|
+
command to run next. No model call, no network. Exit 0 current, 1 stale, 3
|
|
360
|
+
something unknown or a newer model, 2 an error; --strict turns 3 into 1.
|
|
361
|
+
- regrade grades a receipt's saved answers again with another judge (or the
|
|
362
|
+
same judge under a new template): what lib/reuse.js's triage calls
|
|
363
|
+
"regrade". --answers is {"answers": {"<generation_hash>": "<text>"}}; every
|
|
364
|
+
answer is checked against its hash before any call. The generation side of
|
|
365
|
+
the new receipt is the original's, draw for draw; --samples defaults to the
|
|
366
|
+
original's. The receipt records the re-judge (its archived arms, judged_at and
|
|
367
|
+
grader_revision); <receipt>.regrade.json beside it names the original receipt
|
|
368
|
+
and the answers file.
|
|
231
369
|
- export --to summary-json emits the minimal stable interchange summary
|
|
232
370
|
(driftproof/summary v1) other tools can consume without a receipt parser.`);
|
|
233
371
|
}
|
|
234
372
|
|
|
235
|
-
|
|
236
|
-
function dateStamp(iso) { return iso.slice(0, 10); }
|
|
373
|
+
// An imported receipt whose source records no date carries date_utc null (receipt v0.9).
|
|
374
|
+
function dateStamp(iso) { return iso ? iso.slice(0, 10) : 'undated'; }
|
|
237
375
|
|
|
238
376
|
async function cmdRun(positional, flags) {
|
|
239
377
|
const skillDir = positional[0];
|
|
240
378
|
if (!skillDir) { usage(); process.exit(2); }
|
|
379
|
+
refuseUnknownFlags('run', flags);
|
|
380
|
+
refuseExtraPositionals('run', positional, '<skill-dir>');
|
|
241
381
|
|
|
242
382
|
// Per-project defaults from .driftproofrc (if any); CLI flags override these.
|
|
243
383
|
const rc = loadRc(skillDir);
|
|
@@ -279,7 +419,14 @@ async function cmdRun(positional, flags) {
|
|
|
279
419
|
const outDir = flags.out ? path.resolve(flags.out) : RECEIPTS_DIR;
|
|
280
420
|
fs.mkdirSync(outDir, { recursive: true });
|
|
281
421
|
|
|
282
|
-
|
|
422
|
+
let skill;
|
|
423
|
+
try { skill = loadSkillOrRefuse(skillDir, 'run'); } catch (e) {
|
|
424
|
+
// Spec 050 AC-1: a suite whose case ids collide is refused before the projection and
|
|
425
|
+
// before any call, naming the suite file, like the empty suite below.
|
|
426
|
+
if (!e || e.code !== 'DUPLICATE_CASE_ID') throw e;
|
|
427
|
+
console.error(` ✗ REFUSED: the suite at ${path.join(path.resolve(skillDir), 'evals', 'evals.json')}: ${e.message}. Nothing measured, no receipt written.`);
|
|
428
|
+
process.exit(2);
|
|
429
|
+
}
|
|
283
430
|
const nCases = maxCases ? Math.min(maxCases, skill.suite.caseCount) : skill.suite.caseCount;
|
|
284
431
|
// Spec 026 AC-5 (F2): a suite with no cases measures nothing, and a receipt
|
|
285
432
|
// over nothing is a receipt about nothing. Refused here, before the
|
|
@@ -392,10 +539,13 @@ async function cmdRun(positional, flags) {
|
|
|
392
539
|
process.exitCode = 1;
|
|
393
540
|
}
|
|
394
541
|
|
|
395
|
-
|
|
542
|
+
// Spec 062 (register row 3): the name carries the receipt's own hash, so a second
|
|
543
|
+
// run on the same day names a second file, and the file is created, never
|
|
544
|
+
// replaced: a name already taken is a receipt already written.
|
|
545
|
+
const base = receiptBaseName({ skill: skill.name, model: receipt.run.model_id, date: dateStamp(receipt.run.date_utc), hash: receipt.receipt_hash });
|
|
396
546
|
const jsonPath = path.join(outDir, `${base}.json`);
|
|
397
547
|
const mdPath = path.join(outDir, `${base}.summary.md`);
|
|
398
|
-
fs.writeFileSync(jsonPath, JSON.stringify(receipt, null, 2));
|
|
548
|
+
fs.writeFileSync(jsonPath, JSON.stringify(receipt, null, 2), { flag: 'wx' });
|
|
399
549
|
fs.writeFileSync(mdPath, summarizeReceipt(receipt));
|
|
400
550
|
emitted.push(jsonPath);
|
|
401
551
|
|
|
@@ -467,6 +617,122 @@ function cmdDiff(positional, flags) {
|
|
|
467
617
|
}
|
|
468
618
|
}
|
|
469
619
|
|
|
620
|
+
// ── stale (spec 053) ────────────────────────────────────────────────────────
|
|
621
|
+
// Whether each receipt's conclusion still stands under what would run today. Exit 0 current, 1 stale,
|
|
622
|
+
// 3 nothing stale but something unknown or a newer model, 2 an unreadable or invalid receipt or a bad
|
|
623
|
+
// flag; --strict turns 3 into 1. `--json` prints the driftproof.stale/1 document; `--json <path>`
|
|
624
|
+
// writes it there, over an earlier stale document, and refuses (exit 2) a path that holds anything
|
|
625
|
+
// else, so `stale --json receipts/*.json` cannot overwrite a receipt (A-053-1, approval F-1).
|
|
626
|
+
function cmdStale(positional, flags) {
|
|
627
|
+
const { staleReport, renderText } = require('../lib/stale');
|
|
628
|
+
const VALUE = ['skill', 'suite', 'model', 'judge', 'harness-version'];
|
|
629
|
+
const BOOL = ['strict', 'no-harness-check'];
|
|
630
|
+
const bad = (msg) => { console.error(` ✗ stale: ${msg}\n`); console.error(` usage: ${PROJECT_NAME} stale <receipt.json>... [--skill DIR] [--suite FILE] [--model ID] [--judge ID] [--harness-version V | --no-harness-check] [--strict] [--json [path]]`); process.exitCode = 2; };
|
|
631
|
+
for (const k of Object.keys(flags)) if (![...VALUE, ...BOOL, 'json'].includes(k)) return bad(`unknown flag --${k}`);
|
|
632
|
+
for (const k of VALUE) if (flags[k] === true) return bad(`--${k} needs a value`);
|
|
633
|
+
if (flags['harness-version'] && flags['no-harness-check']) return bad('--harness-version and --no-harness-check together');
|
|
634
|
+
const files = [...positional];
|
|
635
|
+
let jsonOut = null;
|
|
636
|
+
if (typeof flags.json === 'string') {
|
|
637
|
+
if (fs.existsSync(flags.json)) {
|
|
638
|
+
let earlier = null; try { earlier = JSON.parse(fs.readFileSync(flags.json, 'utf8')); } catch { /* not a stale document */ }
|
|
639
|
+
if (!earlier || earlier.schema !== 'driftproof.stale/1') return bad(`--json ${flags.json} names an existing file that is not a stale document; nothing is written over it (put --json last with a new path, or give --json alone)`);
|
|
640
|
+
}
|
|
641
|
+
jsonOut = flags.json;
|
|
642
|
+
}
|
|
643
|
+
if (!files.length) return bad('no receipt given');
|
|
644
|
+
if (flags.skill) { try { loadSkill(flags.skill); } catch (e) { return bad(`--skill ${flags.skill}: ${e.message}`); } }
|
|
645
|
+
if (flags.suite) { try { JSON.parse(fs.readFileSync(flags.suite, 'utf8')); } catch (e) { return bad(`--suite ${flags.suite}: ${e.message}`); } }
|
|
646
|
+
const doc = staleReport(files, {
|
|
647
|
+
skill: flags.skill || null, suite: flags.suite || null, model: flags.model || null, judge: flags.judge || null,
|
|
648
|
+
harnessVersion: flags['harness-version'] || null, noHarnessCheck: !!flags['no-harness-check'], strict: !!flags.strict,
|
|
649
|
+
rc: loadRc(flags.skill || null),
|
|
650
|
+
});
|
|
651
|
+
if (flags.json) {
|
|
652
|
+
const text = JSON.stringify(doc, null, 2) + '\n';
|
|
653
|
+
if (jsonOut) { fs.writeFileSync(jsonOut, text); process.stdout.write(renderText(doc)); } else process.stdout.write(text);
|
|
654
|
+
} else process.stdout.write(renderText(doc));
|
|
655
|
+
process.exitCode = doc.exit_code;
|
|
656
|
+
}
|
|
657
|
+
|
|
658
|
+
// ── regrade (spec 044) ──────────────────────────────────────────────────────
|
|
659
|
+
// The executor behind lib/reuse.js's `regrade` decision. Every input is checked
|
|
660
|
+
// before the projection, and the projection before any call: the receipt's seal,
|
|
661
|
+
// the skill and suite it was run on, and every answer against its hash (AC-4).
|
|
662
|
+
async function cmdRegrade(positional, flags) {
|
|
663
|
+
const receiptPath = positional[0];
|
|
664
|
+
if (!receiptPath) { console.error(' ✗ regrade: a receipt path is required\n'); usage(); process.exit(2); }
|
|
665
|
+
refuseUnknownFlags('regrade', flags);
|
|
666
|
+
refuseExtraPositionals('regrade', positional, '<receipt.json>');
|
|
667
|
+
for (const f of ['skill', 'answers', 'judge-model']) {
|
|
668
|
+
if (typeof flags[f] !== 'string' || !flags[f]) { console.error(` ✗ REFUSED (regrade): --${f} is required; nothing run, no receipt written.`); process.exit(2); }
|
|
669
|
+
}
|
|
670
|
+
const checked = checkNumericInputs(flags, {});
|
|
671
|
+
const receipt = JSON.parse(fs.readFileSync(receiptPath, 'utf8'));
|
|
672
|
+
const judgeModel = resolveModel(flags['judge-model']);
|
|
673
|
+
checkRegistered([receipt.run.model_id], () => judgeModel);
|
|
674
|
+
const origSamples = receipt.run && receipt.run.judge && receipt.run.judge.samples;
|
|
675
|
+
const samples = checked.samples !== undefined ? parseInt(checked.samples, 10) : origSamples;
|
|
676
|
+
if (!Number.isInteger(samples) || samples < 1) { console.error(' ✗ REFUSED (regrade): the receipt records no judge sample count; pass --samples.'); process.exit(2); }
|
|
677
|
+
const maxCalls = checked['max-calls'] !== undefined ? parseInt(checked['max-calls'], 10) : DEV_MAX_CALLS;
|
|
678
|
+
const maxUsd = checked['max-usd'] !== undefined ? parseFloat(checked['max-usd']) : DEV_MAX_USD;
|
|
679
|
+
const trusted = !!flags['trusted-skill'];
|
|
680
|
+
const outDir = flags.out ? path.resolve(flags.out) : RECEIPTS_DIR;
|
|
681
|
+
const skill = loadSkillOrRefuse(flags.skill, 'regrade');
|
|
682
|
+
const answersFile = path.resolve(flags.answers);
|
|
683
|
+
const answersBytes = fs.readFileSync(answersFile);
|
|
684
|
+
const answers = (JSON.parse(answersBytes.toString('utf8')) || {}).answers || {};
|
|
685
|
+
|
|
686
|
+
const plan = planRegrade({ receipt, skill, answers, judgeModel, samples });
|
|
687
|
+
console.log(`\n${PROJECT_NAME} regrade: ${path.basename(receiptPath)}`);
|
|
688
|
+
if (process.env.DRIFTPROOF_STUB === '1') console.log(' STUB RUN (DRIFTPROOF_STUB=1): nothing will answer the judge; the receipt will be UNVERIFIED and grade nothing');
|
|
689
|
+
console.log(` model: ${receipt.run.model_id} (generations kept) judge: ${receipt.run.judge.model_id} → ${judgeModel} samples/answer: ${samples}`);
|
|
690
|
+
if (plan.problems.length) {
|
|
691
|
+
console.error(`\n ✗ REFUSED (regrade): ${plan.problems.length} problem(s) with the inputs; nothing run, no receipt written.`);
|
|
692
|
+
for (const p of plan.problems.slice(0, 20)) console.error(` - ${p}`);
|
|
693
|
+
process.exit(2);
|
|
694
|
+
}
|
|
695
|
+
console.log(` answers to grade: ${plan.draws} projected calls: ${plan.calls} cap: ${maxCalls}`);
|
|
696
|
+
console.log(` projected cost: ~$${plan.usd.toFixed(2)} (the runner's per-call judge estimate; budget $${maxUsd.toFixed(2)}, hard-stop $${(maxUsd * 1.25).toFixed(2)})`);
|
|
697
|
+
if (plan.calls > maxCalls) {
|
|
698
|
+
console.error(`\n ✗ ABORT (cost guard): projected ${plan.calls} calls exceeds --max-calls ${maxCalls}.`);
|
|
699
|
+
process.exit(3);
|
|
700
|
+
}
|
|
701
|
+
if (plan.usd > maxUsd) {
|
|
702
|
+
console.error(`\n ✗ ABORT (cost guard): projected cost ~$${plan.usd.toFixed(2)} exceeds --max-usd $${maxUsd.toFixed(2)}.`);
|
|
703
|
+
process.exit(3);
|
|
704
|
+
}
|
|
705
|
+
let result;
|
|
706
|
+
try {
|
|
707
|
+
result = await regradeReceipt({
|
|
708
|
+
receipt, skill, answers, judgeModel, samples,
|
|
709
|
+
opts: {
|
|
710
|
+
trusted, budget: new BudgetTracker(maxUsd),
|
|
711
|
+
onProgress: (p) => { if (p.phase === 'done') console.log(` ${p.case} / ${p.mode}: ${p.outcome} (${p.score.toFixed(2)} ± ${(p.stddev || 0).toFixed(2)})`); },
|
|
712
|
+
},
|
|
713
|
+
});
|
|
714
|
+
} catch (e) {
|
|
715
|
+
if (e && e.code === 'SUBSTRATE_MISMATCH') { console.error(`\n ✗ REFUSED (substrate mismatch): ${e.message}`); process.exit(5); }
|
|
716
|
+
if (e && e.code === 'BUDGET_HARDSTOP') { console.error(`\n ✗ ABORT (budget hard-stop): ${e.message}`); process.exit(3); }
|
|
717
|
+
throw e;
|
|
718
|
+
}
|
|
719
|
+
const out = result.receipt;
|
|
720
|
+
const { valid, errors } = validateReceipt(out);
|
|
721
|
+
if (!valid) { console.error(' ✗ regraded receipt FAILED schema validation:', JSON.stringify(errors, null, 2)); process.exitCode = 1; }
|
|
722
|
+
if (!verifyReceiptHash(out)) { console.error(' ✗ receipt_hash does not verify'); process.exitCode = 1; }
|
|
723
|
+
fs.mkdirSync(outDir, { recursive: true });
|
|
724
|
+
const base = receiptBaseName({ skill: out.skill.name, model: out.run.model_id, tag: `regrade-${fileSlug(judgeModel)}`, date: dateStamp(out.run.date_utc), hash: out.receipt_hash });
|
|
725
|
+
const jsonPath = path.join(outDir, `${base}.json`);
|
|
726
|
+
fs.writeFileSync(jsonPath, JSON.stringify(out, null, 2), { flag: 'wx' });
|
|
727
|
+
fs.writeFileSync(path.join(outDir, `${base}.summary.md`), summarizeReceipt(out));
|
|
728
|
+
const provenance = { ...result.provenance, answers: { file: path.basename(answersFile), sha256: sha256(answersBytes), graded: plan.draws } };
|
|
729
|
+
// The sidecar's name is lib/receipt.js's, which the directory readers skip by (spec 069).
|
|
730
|
+
const sidecar = `${base}${REGRADE_SIDECAR.suffix}`;
|
|
731
|
+
fs.writeFileSync(path.join(outDir, sidecar), JSON.stringify(provenance, null, 2));
|
|
732
|
+
console.log(` → ${path.relative(process.cwd(), jsonPath)} (${result.calls} judge calls)`);
|
|
733
|
+
console.log(` → verification_level ${out.verification_level}; provenance: ${sidecar}\n`);
|
|
734
|
+
}
|
|
735
|
+
|
|
470
736
|
function cmdValidate(positional) {
|
|
471
737
|
const p = positional[0];
|
|
472
738
|
if (!p) { usage(); process.exit(2); }
|
|
@@ -515,6 +781,8 @@ function cmdBadge(positional, flags) {
|
|
|
515
781
|
if (fs.existsSync(p) && fs.statSync(p).isDirectory()) return cmdBadgeSet(p, flags);
|
|
516
782
|
const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
517
783
|
refuseUnverified(receipt, p, 'badge');
|
|
784
|
+
refuseAmbiguous(receipt, p, 'badge');
|
|
785
|
+
refuseInvalid(receipt, p, 'badge');
|
|
518
786
|
if (flags.svg) {
|
|
519
787
|
// SPEC 036. The drawn badge: state, model, date, lift and uncertainty, with the
|
|
520
788
|
// machine token in its data attributes. After the same hash check as the JSON.
|
|
@@ -552,7 +820,8 @@ function cmdBadge(positional, flags) {
|
|
|
552
820
|
// read; when --models is given the requested list governs, so a model with no
|
|
553
821
|
// receipt is `refused` and can be the worst state.
|
|
554
822
|
function cmdBadgeSet(dir, flags) {
|
|
555
|
-
const
|
|
823
|
+
const listed = listReceipts(dir);
|
|
824
|
+
const files = listed.map((l) => l.file);
|
|
556
825
|
if (files.length === 0) {
|
|
557
826
|
console.error(` \u2717 REFUSED (badge): ${dir} holds no receipts; a badge over nothing is a badge about nothing.`);
|
|
558
827
|
process.exit(2);
|
|
@@ -560,13 +829,17 @@ function cmdBadgeSet(dir, flags) {
|
|
|
560
829
|
// Every receipt is verified before it is displayed, exactly as the
|
|
561
830
|
// single-receipt path does (spec 026 AC-20): a set is not a way around the
|
|
562
831
|
// check, and the file that fails is named.
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
832
|
+
// An unreadable file is left to the decision, which fails its model closed and
|
|
833
|
+
// names it (spec 030 AC-2), as `decide` does.
|
|
834
|
+
for (const { file, receipt } of listed) {
|
|
835
|
+
if (!receipt) continue;
|
|
836
|
+
refuseUnverified(receipt, file, 'badge');
|
|
837
|
+
// Spec 062: validated too. A receipt with duplicate rows is left to the decision's
|
|
838
|
+
// `refused` state (spec 050 AC-3), which fails closed.
|
|
839
|
+
if (!duplicateCaseRows(receipt).length) refuseInvalid(receipt, file, 'badge');
|
|
566
840
|
}
|
|
567
|
-
const models = flags.models ||
|
|
568
|
-
.map((
|
|
569
|
-
.map((r) => r.run && r.run.model_id).filter(Boolean).join(',');
|
|
841
|
+
const models = flags.models || listed
|
|
842
|
+
.map((l) => l.receipt && l.receipt.run && l.receipt.run.model_id).filter(Boolean).join(',');
|
|
570
843
|
const d = decision.decideSet(dir, models);
|
|
571
844
|
if (flags['github-output']) { console.log(decision.githubOutputLines(d)); return; }
|
|
572
845
|
const badge = decision.badgeEndpointForSet(d);
|
|
@@ -619,10 +892,11 @@ function cmdDecide(positional, flags) {
|
|
|
619
892
|
// is unreadable and fails closed on both, naming which (spec 030 AC-2,
|
|
620
893
|
// absence-vs-unreadable). Refusing it here would collapse that distinction
|
|
621
894
|
// into an exit code.
|
|
622
|
-
for (const
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
|
|
895
|
+
for (const { file, receipt } of listReceipts(dir)) {
|
|
896
|
+
if (!receipt) continue;
|
|
897
|
+
refuseUnverified(receipt, file, 'decide');
|
|
898
|
+
// Spec 062 (register row 2): and validated, as badge is.
|
|
899
|
+
if (!duplicateCaseRows(receipt).length) refuseInvalid(receipt, file, 'decide');
|
|
626
900
|
}
|
|
627
901
|
const failOnRegression = String(flags['fail-on-regression'] ?? 'true') !== 'false';
|
|
628
902
|
const d = decision.decideSet(dir, flags.models, { failOnRegression });
|
|
@@ -661,12 +935,27 @@ function cmdImport(positional, flags) {
|
|
|
661
935
|
const p = positional[0];
|
|
662
936
|
const from = flags.from;
|
|
663
937
|
const { IMPORT_TOOLS, importResults } = require('../lib/importers');
|
|
938
|
+
const { ANTHROPIC_TOOLS, importAnthropic, ImportRefused } = require('../lib/importers-anthropic');
|
|
939
|
+
const tools = [...IMPORT_TOOLS, ...ANTHROPIC_TOOLS];
|
|
664
940
|
if (!p || typeof from !== 'string') {
|
|
665
|
-
console.error(`usage: driftproof import <results.json> --from ${
|
|
941
|
+
console.error(`usage: driftproof import <results.json|dir> --from ${tools.join('|')} [--count-errored-runs] [--out DIR]`);
|
|
666
942
|
process.exit(2);
|
|
667
943
|
}
|
|
668
|
-
|
|
669
|
-
|
|
944
|
+
let receipt;
|
|
945
|
+
let read = p;
|
|
946
|
+
if (ANTHROPIC_TOOLS.includes(from)) {
|
|
947
|
+
// Spec 049: a file or a directory; the clock is read once, into run.import.imported_at.
|
|
948
|
+
try {
|
|
949
|
+
({ receipt, file: read } = importAnthropic(p, { from, importedAt: new Date().toISOString(), countErroredRuns: flags['count-errored-runs'] === true }));
|
|
950
|
+
} catch (e) {
|
|
951
|
+
if (!(e instanceof ImportRefused)) throw e;
|
|
952
|
+
console.error(`✗ import refused, nothing written: ${e.message}`);
|
|
953
|
+
process.exit(1);
|
|
954
|
+
}
|
|
955
|
+
} else {
|
|
956
|
+
const data = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
957
|
+
receipt = importResults(data, { from, importedAt: new Date().toISOString() });
|
|
958
|
+
}
|
|
670
959
|
const { valid, errors } = validateReceipt(receipt);
|
|
671
960
|
if (!valid || !verifyReceiptHash(receipt)) {
|
|
672
961
|
console.error('✗ converted receipt failed validation — nothing written:', JSON.stringify(errors, null, 2));
|
|
@@ -674,30 +963,43 @@ function cmdImport(positional, flags) {
|
|
|
674
963
|
}
|
|
675
964
|
const outDir = flags.out ? path.resolve(flags.out) : RECEIPTS_DIR;
|
|
676
965
|
fs.mkdirSync(outDir, { recursive: true });
|
|
677
|
-
|
|
966
|
+
// Spec 049 (approval F-2): the two formats it adds produce several documents per skill, model and
|
|
967
|
+
// day (one per iteration or per results directory), so their receipt name carries the document's
|
|
968
|
+
// own hash and one import never overwrites another's receipt. The same document names the same file.
|
|
969
|
+
// The same base-name rule as a run's (spec 062), with the source document's hash in
|
|
970
|
+
// the receipt hash's place, so the name spec 049 gave an import is unchanged.
|
|
971
|
+
const doc = receipt.run.import && ANTHROPIC_TOOLS.includes(from) ? receipt.run.import.source_sha256 : null;
|
|
972
|
+
const base = receiptBaseName({ skill: receipt.skill.name, model: receipt.run.model_id, tag: `imported-${fileSlug(from)}`, date: dateStamp(receipt.run.date_utc), hash: doc });
|
|
678
973
|
const jsonPath = path.join(outDir, `${base}.json`);
|
|
679
974
|
fs.writeFileSync(jsonPath, JSON.stringify(receipt, null, 2));
|
|
975
|
+
if (read !== p) console.log(`read ${path.relative(process.cwd(), read)}`);
|
|
680
976
|
console.log(`imported → ${path.relative(process.cwd(), jsonPath)}`);
|
|
681
977
|
console.log(` verification_level: DECLARED (source ${receipt.run.source}) — the source tool's declaration, converted faithfully;`);
|
|
682
978
|
console.log(` no generation hashes (never fabricated); excluded from drift verdicts unless re-run TESTED. See docs/interop.md.`);
|
|
979
|
+
for (const n of (receipt.run.import && receipt.run.import.notices) || []) console.log(` notice: ${n}`);
|
|
683
980
|
}
|
|
684
981
|
|
|
685
982
|
// Emit the minimal stable interchange summary for one receipt (interop, Phase 7).
|
|
686
983
|
function cmdExport(positional, flags) {
|
|
687
984
|
const p = positional[0];
|
|
688
985
|
const to = flags.to || 'summary-json';
|
|
689
|
-
if (!p) { console.error('usage: driftproof export <receipt.json> [--to summary-json] [--report-url URL] [--out FILE]'); process.exit(2); }
|
|
986
|
+
if (!p) { console.error('usage: driftproof export <receipt.json> [--to summary-json] [--report-url URL] [--out FILE|DIR]'); process.exit(2); }
|
|
690
987
|
if (to !== 'summary-json') { console.error(`unknown export target "${to}" — supported: summary-json`); process.exit(2); }
|
|
691
988
|
const { toSummaryJson } = require('../lib/export');
|
|
692
989
|
const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
693
990
|
refuseUnverified(receipt, p, 'export');
|
|
991
|
+
refuseAmbiguous(receipt, p, 'export');
|
|
694
992
|
const summary = toSummaryJson(receipt, { reportUrl: typeof flags['report-url'] === 'string' ? flags['report-url'] : null });
|
|
695
993
|
const json = JSON.stringify(summary, null, 2);
|
|
696
994
|
if (flags.out) {
|
|
697
|
-
|
|
995
|
+
// An existing directory gets the receipt's base name with the summary suffix, the name
|
|
996
|
+
// listReceipts skips beside its receipt (spec 107); any other path is written as given.
|
|
997
|
+
let shown = flags.out;
|
|
998
|
+
if (fs.existsSync(path.resolve(shown)) && fs.statSync(path.resolve(shown)).isDirectory()) shown = path.join(shown, path.basename(p, '.json') + SUMMARY_SIDECAR.suffix);
|
|
999
|
+
const out = path.resolve(shown);
|
|
698
1000
|
fs.mkdirSync(path.dirname(out), { recursive: true });
|
|
699
1001
|
fs.writeFileSync(out, json + '\n');
|
|
700
|
-
console.log(`summary written to ${
|
|
1002
|
+
console.log(`summary written to ${shown} (${summary.verdict}, delta ${summary.delta === null ? 'n/a' : summary.delta})`);
|
|
701
1003
|
} else {
|
|
702
1004
|
console.log(json);
|
|
703
1005
|
}
|
|
@@ -710,6 +1012,8 @@ async function main() {
|
|
|
710
1012
|
case 'init': return cmdInit(positional);
|
|
711
1013
|
case 'run': return cmdRun(positional, flags);
|
|
712
1014
|
case 'diff': return cmdDiff(positional, flags);
|
|
1015
|
+
case 'regrade': return cmdRegrade(positional, flags);
|
|
1016
|
+
case 'stale': return cmdStale(positional, flags);
|
|
713
1017
|
case 'validate': return cmdValidate(positional);
|
|
714
1018
|
case 'badge': return cmdBadge(positional, flags);
|
|
715
1019
|
case 'decide': return cmdDecide(positional, flags);
|
package/config/models.json
CHANGED
|
@@ -10,8 +10,9 @@
|
|
|
10
10
|
"models": [
|
|
11
11
|
{ "id": "claude-fable-5-1", "family": "fable", "provider": "anthropic", "released": "2026-09-01", "input_price": 10.0, "output_price": 50.0, "cache_read_price": 0.25, "tier": "frontier", "judge_eligible": false, "surface_verified": "claude-cli", "surface_verified_at": "2026-09-01" },
|
|
12
12
|
{ "id": "claude-fable-5", "family": "fable", "provider": "anthropic", "released": "2026-06-09", "input_price": 10.0, "output_price": 50.0, "tier": "frontier", "judge_eligible": false, "lifecycle": "legacy" },
|
|
13
|
-
{ "id": "claude-opus-5",
|
|
14
|
-
{ "id": "claude-opus-
|
|
13
|
+
{ "id": "claude-opus-5-5", "family": "opus", "provider": "anthropic", "released": "2026-09-22", "input_price": 4.0, "output_price": 20.0, "cache_read_price": 0.2, "tier": "frontier", "judge_eligible": false },
|
|
14
|
+
{ "id": "claude-opus-5", "family": "opus", "provider": "anthropic", "released": "2026-07-24", "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
|
|
15
|
+
{ "id": "claude-opus-4-8", "family": "opus", "provider": "anthropic", "released": "2026-05-28", "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
|
|
15
16
|
{ "id": "claude-opus-4-7", "family": "opus", "provider": "anthropic", "released": null, "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
|
|
16
17
|
{ "id": "claude-opus-4-6", "family": "opus", "provider": "anthropic", "released": null, "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
|
|
17
18
|
{ "id": "claude-opus-4-5-20251101", "family": "opus", "provider": "anthropic", "released": "2025-11-01", "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
|
package/config.js
CHANGED
|
@@ -9,7 +9,7 @@ const PROJECT_NAME = 'driftproof';
|
|
|
9
9
|
// Bumped whenever the runner's behaviour or receipt-generation semantics change
|
|
10
10
|
// in a way that could affect results. Recorded into every receipt as
|
|
11
11
|
// run.runner_version so a receipt is reproducible against a known engine.
|
|
12
|
-
const RUNNER_VERSION = '0.
|
|
12
|
+
const RUNNER_VERSION = '0.12.0';
|
|
13
13
|
|
|
14
14
|
// The eval format we CONSUME (we deliberately do not invent our own).
|
|
15
15
|
const SUITE_FORMAT = 'agentskills.io/evals';
|
|
@@ -38,7 +38,7 @@ const SUITE_FORMAT = 'agentskills.io/evals';
|
|
|
38
38
|
// results.aggregates.band_rule, and bands that are null where the formula
|
|
39
39
|
// cannot form. NOT additive for the validator: a v0.5 receipt restamped 0.6
|
|
40
40
|
// is refused. v0.5 is frozen as receipt.v0.5.schema.json.
|
|
41
|
-
const RECEIPT_SCHEMA_VERSION = '0.
|
|
41
|
+
const RECEIPT_SCHEMA_VERSION = '0.9';
|
|
42
42
|
|
|
43
43
|
// Input bounds on the skill loader and the post-checks (spec 026 AC-13, audit
|
|
44
44
|
// A3/A7). lib/skill.js loaded every bundled file with no count, depth, byte,
|