driftproof 0.11.3 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +53 -3
- package/bin/driftproof +159 -41
- package/config.js +1 -1
- package/lib/decision.js +33 -25
- package/lib/export.js +2 -1
- package/lib/init.js +3 -2
- package/lib/receipt.js +111 -8
- package/lib/regrade.js +2 -2
- package/lib/skill.js +44 -6
- package/lib/verdict.js +33 -17
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -256,7 +256,9 @@ A skill directory is expected to look like:
|
|
|
256
256
|
my-skill/
|
|
257
257
|
SKILL.md # the skill instructions (required)
|
|
258
258
|
evals/evals.json # agentskills.io/evals suite (required)
|
|
259
|
-
.driftproofrc # optional
|
|
259
|
+
.driftproofrc # optional run defaults (models, max_usd); samples, max_cases
|
|
260
|
+
# and judge_model are read only from the working directory's rc
|
|
261
|
+
# (the GitHub Action reads no working-directory rc at all)
|
|
260
262
|
... # any bundled files (contribute to content_hash)
|
|
261
263
|
```
|
|
262
264
|
|
|
@@ -332,7 +334,7 @@ A receipt is the unit of evidence — one JSON document conforming to
|
|
|
332
334
|
"model_release_date": "2025-10-01",
|
|
333
335
|
"provider": "anthropic",
|
|
334
336
|
"surface": "claude-cli",
|
|
335
|
-
"runner_version": "0.
|
|
337
|
+
"runner_version": "0.12.0",
|
|
336
338
|
"date_utc": "2026-07-27T…Z",
|
|
337
339
|
"registry": "registered",
|
|
338
340
|
"transcripts": "hashes-only",
|
|
@@ -411,7 +413,7 @@ jobs:
|
|
|
411
413
|
runs-on: ubuntu-latest
|
|
412
414
|
steps:
|
|
413
415
|
- uses: actions/checkout@v4
|
|
414
|
-
- uses: driftproofhq/driftproof@v0.
|
|
416
|
+
- uses: driftproofhq/driftproof@v0.12.0
|
|
415
417
|
with:
|
|
416
418
|
skill-dir: skills/my-skill
|
|
417
419
|
models: claude-haiku-4-5
|
|
@@ -427,6 +429,13 @@ model, uploads the whole receipt directory as a build artifact, and renders the
|
|
|
427
429
|
run on three surfaces: the **badge**, a **job summary** carrying one row per
|
|
428
430
|
requested model, and the **check title** of the enforcement step.
|
|
429
431
|
|
|
432
|
+
**The checkout's `.driftproofrc` is not read.** The action starts the run in an
|
|
433
|
+
empty working directory, so no key of a `.driftproofrc` at the repository root
|
|
434
|
+
is read, and a skill directory's `.driftproofrc` may not set `max_cases`,
|
|
435
|
+
`samples` or `judge_model`. Both files are pull-request content, and a pull
|
|
436
|
+
request may not narrow the run that measures it. The run takes its models and
|
|
437
|
+
caps from the inputs above.
|
|
438
|
+
|
|
430
439
|
**The decision is taken over every receipt the run produced.** Each requested
|
|
431
440
|
model gets one decision state, and the run's `verdict` is the **worst** of them —
|
|
432
441
|
worst first, in this order:
|
|
@@ -490,6 +499,47 @@ surface. The CLI has carried its own input contract at its own door since
|
|
|
490
499
|
`npx driftproof` refuses a malformed cap or model id before it projects a run.
|
|
491
500
|
Neither statement covers the other. Said of the Action, of CI, or of the runner, "hostile input is refused" is only ever true of the one surface it was measured on.
|
|
492
501
|
|
|
502
|
+
### Scheduled stale check
|
|
503
|
+
|
|
504
|
+
`driftproof stale` says whether each receipt's conclusion still stands under the model, harness,
|
|
505
|
+
skill, suite and judge that would run today. The staleness check runs it on a schedule in your
|
|
506
|
+
repository. While any receipt needs a rerun or a regrade, one issue labelled `driftproof-stale` lists
|
|
507
|
+
each one, what moved, and the command to run next. Later runs update that issue, and the first run
|
|
508
|
+
that finds everything current closes it. It makes no model call and needs no API key.
|
|
509
|
+
|
|
510
|
+
Copy [`examples/workflows/driftproof-stale.yml`](examples/workflows/driftproof-stale.yml) into
|
|
511
|
+
`.github/workflows/`. It runs weekly and on demand, with these permissions and no others:
|
|
512
|
+
|
|
513
|
+
```yaml
|
|
514
|
+
permissions:
|
|
515
|
+
contents: read
|
|
516
|
+
issues: write
|
|
517
|
+
# ...
|
|
518
|
+
- uses: driftproofhq/driftproof/stale@v0.12.0
|
|
519
|
+
with:
|
|
520
|
+
receipts: receipts/**/*.json
|
|
521
|
+
skill: skills/my-skill
|
|
522
|
+
```
|
|
523
|
+
|
|
524
|
+
| Input | Default | What it does |
|
|
525
|
+
|---|---|---|
|
|
526
|
+
| `receipts` | `receipts/**/*.json` | Glob of receipt files, one per line for several. |
|
|
527
|
+
| `skill` | none | The skill directory as it is today. Without it, the skill axis reads unknown. |
|
|
528
|
+
| `suite` | the skill's `evals/evals.json` | The eval suite file. |
|
|
529
|
+
| `model`, `judge` | `.driftproofrc` | The model and the judge that would run today. |
|
|
530
|
+
| `harness-version` | `latest` | A Claude Code version, `latest` (read from npm), or `none` (not checked). |
|
|
531
|
+
| `strict` | `false` | Count an unknown axis or an advisory as stale. |
|
|
532
|
+
| `fail-on-stale` | `false` | Fail the job while anything is stale. |
|
|
533
|
+
| `open-issue` | `true` | Keep the issue. |
|
|
534
|
+
| `issue-label` | `driftproof-stale` | Give each check its own label to keep separate issues. |
|
|
535
|
+
| `github-token` | the workflow's token | Used for the issue only. |
|
|
536
|
+
|
|
537
|
+
The job fails on an error, such as a receipt that does not validate, and on stale only when
|
|
538
|
+
`fail-on-stale` is `'true'`. A patch release of the harness alone (2.1.280 to 2.1.281) is an
|
|
539
|
+
advisory: the job summary reports it and nothing fails. An axis that cannot be known is reported as
|
|
540
|
+
unknown, never as current. Outputs: `result` (`current`, `advisory`, `stale` or `error`),
|
|
541
|
+
`stale-count`, `issue-number` and `report-dir`.
|
|
542
|
+
|
|
493
543
|
### Badge
|
|
494
544
|
|
|
495
545
|
`driftproof badge <receipt>` emits a [shields.io endpoint](https://shields.io/badges/endpoint-badge)
|
package/bin/driftproof
CHANGED
|
@@ -8,7 +8,7 @@ const { PROJECT_NAME, RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES, DEV_MAX_USD, DEV_MA
|
|
|
8
8
|
const { loadSkill } = require('../lib/skill');
|
|
9
9
|
const { runSkillOnModel, summarizeReceipt, projectCalls, answeredLine, band, uncertaintyStr } = require('../lib/run');
|
|
10
10
|
const { SAMPLING } = require('../lib/sampling');
|
|
11
|
-
const { validateReceipt, verifyReceiptHash, duplicateCaseRows, ambiguityLine } = require('../lib/receipt');
|
|
11
|
+
const { validateReceipt, verifyReceiptHash, duplicateCaseRows, ambiguityLine, listReceipts, receiptBaseName, fileSlug, REGRADE_SIDECAR, SUMMARY_SIDECAR } = require('../lib/receipt');
|
|
12
12
|
const { buildDriftReport, revisionPairProblem } = require('../lib/diff');
|
|
13
13
|
const { surfaceForModel, isSubscriptionSurface, resolveModel, evalUser } = require('../lib/provider');
|
|
14
14
|
const { estimateRunCostUSD, BudgetTracker } = require('../lib/cost');
|
|
@@ -33,7 +33,7 @@ function writeTranscripts(receipt, transcripts) {
|
|
|
33
33
|
const manifest = { receipt_hash: receipt.receipt_hash, model_id: receipt.run.model_id, date_utc: receipt.run.date_utc, entries: [] };
|
|
34
34
|
for (const t of transcripts) {
|
|
35
35
|
if (!t) continue;
|
|
36
|
-
const base = `${
|
|
36
|
+
const base = `${fileSlug(t.id)}-${t.mode}`;
|
|
37
37
|
fs.writeFileSync(path.join(dir, `${base}.json`), JSON.stringify({ id: t.id, mode: t.mode, generation: t.generation, judge_outputs: t.judge_outputs }, null, 2));
|
|
38
38
|
manifest.entries.push(`${base}.json`);
|
|
39
39
|
}
|
|
@@ -75,24 +75,67 @@ function refuseAmbiguous(receipt, file, command) {
|
|
|
75
75
|
process.exit(2);
|
|
76
76
|
}
|
|
77
77
|
|
|
78
|
-
//
|
|
79
|
-
//
|
|
80
|
-
//
|
|
78
|
+
// ── A RECEIPT THAT DOES NOT VALIDATE IS NOT RENDERED (spec 062, register row 2) ─
|
|
79
|
+
// badge and decide verified the hash and never validated, so a sealed receipt whose
|
|
80
|
+
// schema_version named an Object.prototype property rendered `passing` and decided
|
|
81
|
+
// `helped`. Called after refuseUnverified, and after refuseAmbiguous where a surface
|
|
82
|
+
// calls that, so spec 050's exit 2 for an ambiguous receipt stands: nothing on
|
|
83
|
+
// stdout, no file, the reason on stderr naming the file, exit 4 as for a receipt
|
|
84
|
+
// whose hash does not verify.
|
|
85
|
+
function refuseInvalid(receipt, file, command) {
|
|
86
|
+
const { valid, errors } = validateReceipt(receipt);
|
|
87
|
+
if (valid) return;
|
|
88
|
+
const first = errors[0] || {};
|
|
89
|
+
console.error(` ✗ REFUSED (${command}): ${path.basename(file)} does not validate against its schema (${errors.length} error(s); ${first.instancePath || '/'} ${first.message || ''}); nothing rendered. Run driftproof validate on it for the list.`);
|
|
90
|
+
process.exit(4);
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
// Read an optional .driftproofrc (JSON) for per-project run defaults: the CWD's,
|
|
94
|
+
// then the skill dir's (where `driftproof init` writes it). CLI flags always win
|
|
95
|
+
// over the rc; the rc wins over built-in defaults.
|
|
96
|
+
//
|
|
97
|
+
// Spec 062 (register row 6): an rc that does not parse is REFUSED, exit 2, naming
|
|
98
|
+
// the file. It used to be ignored, so a trailing comma ran the project with the
|
|
99
|
+
// built-in budget and caps instead of the ones the file set.
|
|
100
|
+
//
|
|
101
|
+
// Spec 062 (register row 4): the skill directory is the skill's content (in the
|
|
102
|
+
// Action, pull-request content), and it may not narrow the run that measures it.
|
|
103
|
+
// max_cases, samples and judge_model are read from the working directory's rc
|
|
104
|
+
// only; set in the skill directory's they are ignored and named on stderr. When
|
|
105
|
+
// the working directory IS the skill directory, its rc is the skill's.
|
|
106
|
+
//
|
|
107
|
+
// Spec 069: in the Action the checkout's root is pull-request content too, so
|
|
108
|
+
// action/run.sh starts `run` from an empty directory and no working-directory
|
|
109
|
+
// rc is read there at all.
|
|
110
|
+
const SKILL_RC_IGNORED = ['max_cases', 'samples', 'judge_model'];
|
|
111
|
+
function readRc(p) {
|
|
112
|
+
if (!fs.existsSync(p)) return {};
|
|
113
|
+
let rc;
|
|
114
|
+
try { rc = JSON.parse(fs.readFileSync(p, 'utf8')); } catch (e) {
|
|
115
|
+
console.error(` ✗ REFUSED: ${p} is not valid JSON (${e.message}); nothing run. Fix the file or remove it.`);
|
|
116
|
+
process.exit(2);
|
|
117
|
+
}
|
|
118
|
+
if (!rc || typeof rc !== 'object' || Array.isArray(rc)) {
|
|
119
|
+
console.error(` ✗ REFUSED: ${p} is not a JSON object; nothing run. Fix the file or remove it.`);
|
|
120
|
+
process.exit(2);
|
|
121
|
+
}
|
|
122
|
+
return rc;
|
|
123
|
+
}
|
|
81
124
|
function loadRc(skillDir) {
|
|
82
125
|
const merged = {};
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
126
|
+
const skillRcDir = skillDir ? path.resolve(skillDir) : null;
|
|
127
|
+
if (path.resolve(process.cwd()) !== skillRcDir) Object.assign(merged, readRc(path.join(process.cwd(), '.driftproofrc')));
|
|
128
|
+
if (skillRcDir) {
|
|
129
|
+
const p = path.join(skillRcDir, '.driftproofrc');
|
|
130
|
+
const rc = readRc(p);
|
|
131
|
+
const dropped = SKILL_RC_IGNORED.filter((k) => Object.hasOwn(rc, k));
|
|
132
|
+
for (const k of dropped) delete rc[k];
|
|
133
|
+
if (dropped.length) console.error(` note: ${p} sets ${dropped.join(', ')}; ignored, because a skill directory may not narrow the run that measures it (set them as flags, or in the working directory's .driftproofrc when you run the CLI yourself; the GitHub Action reads no working-directory .driftproofrc)`);
|
|
134
|
+
Object.assign(merged, rc);
|
|
88
135
|
}
|
|
89
136
|
return merged;
|
|
90
137
|
}
|
|
91
138
|
|
|
92
|
-
// Flags that never take a value, so `--trusted-skill <skill-dir>` keeps the dir
|
|
93
|
-
// positional instead of swallowing it as the flag's value.
|
|
94
|
-
const BOOLEAN_FLAGS = new Set(['trusted-skill', 'svg', 'count-errored-runs', 'strict', 'no-harness-check']);
|
|
95
|
-
|
|
96
139
|
// ── THE INPUT CONTRACT (spec 026 AC-9, F5) ────────────────────────────────────
|
|
97
140
|
// A COPY of action/lib.sh's RE_* rules, in the shape spec 028's fixture holds
|
|
98
141
|
// them; specs/026-receipt-integrity/fixtures/input-contract.json carries the
|
|
@@ -168,21 +211,75 @@ function checkNumericInputs(flags, rc) {
|
|
|
168
211
|
return out;
|
|
169
212
|
}
|
|
170
213
|
|
|
214
|
+
// Spec 062 (register row 6): `--name=value` is the flag `name` with the value
|
|
215
|
+
// `value`, the empty value included. It used to be a flag named `name=value`, which
|
|
216
|
+
// no command read, so `--max-usd=0.01` ran with the default budget.
|
|
217
|
+
//
|
|
218
|
+
// A flag that takes no value is set by its name alone, so `--trusted-skill <skill-dir>`
|
|
219
|
+
// keeps the dir positional instead of swallowing it as the flag's value (spec 106: all
|
|
220
|
+
// eight, where three used to take the next argument). Given with `=`, or followed by a
|
|
221
|
+
// boolean word, it is refused, exit 2, naming it, for every command (A-062-3, spec 106):
|
|
222
|
+
// `--trusted-skill=false` and `--trusted-skill false` each turned the same-user lane on.
|
|
223
|
+
const NO_VALUE_FLAGS = new Set(['trusted-skill', 'svg', 'count-errored-runs', 'strict', 'no-harness-check', 'keep-transcripts', 'github-output', 'enforce']);
|
|
224
|
+
const BOOLEAN_WORD = /^(true|false|yes|no|on|off|1|0)$/i;
|
|
225
|
+
function refuseValue(key, given) {
|
|
226
|
+
console.error(` \u2717 REFUSED: --${key} takes no value, and was given one (${given}). Give --${key} alone to set it, or leave it out. Nothing run, no receipt written.`);
|
|
227
|
+
process.exit(2);
|
|
228
|
+
}
|
|
171
229
|
function parseArgs(argv) {
|
|
172
230
|
const positional = [];
|
|
173
231
|
const flags = {};
|
|
174
232
|
for (let i = 0; i < argv.length; i++) {
|
|
175
233
|
const a = argv[i];
|
|
234
|
+
const eq = a.startsWith('--') ? a.indexOf('=') : -1;
|
|
235
|
+
if (eq > 2 && NO_VALUE_FLAGS.has(a.slice(2, eq))) refuseValue(a.slice(2, eq), a);
|
|
236
|
+
if (eq > 2) { flags[a.slice(2, eq)] = a.slice(eq + 1); continue; }
|
|
176
237
|
if (a.startsWith('--')) {
|
|
177
238
|
const key = a.slice(2);
|
|
178
239
|
const next = argv[i + 1];
|
|
179
|
-
if (
|
|
240
|
+
if (NO_VALUE_FLAGS.has(key) && next !== undefined && BOOLEAN_WORD.test(next)) refuseValue(key, `${a} ${next}`);
|
|
241
|
+
if (NO_VALUE_FLAGS.has(key) || next === undefined || next.startsWith('--')) { flags[key] = true; }
|
|
180
242
|
else { flags[key] = next; i++; }
|
|
181
243
|
} else positional.push(a);
|
|
182
244
|
}
|
|
183
245
|
return { positional, flags };
|
|
184
246
|
}
|
|
185
247
|
|
|
248
|
+
// Spec 062 (register row 6): the two commands that spend refuse a flag they do not
|
|
249
|
+
// take, exit 2, naming it, before anything is read or printed. A misspelt cap flag
|
|
250
|
+
// used to be read by nothing, and the run went ahead with the default cap. The
|
|
251
|
+
// `node:util` parseArgs rewrite, one table per command, is later work.
|
|
252
|
+
const COMMAND_FLAGS = {
|
|
253
|
+
run: ['models', 'samples', 'max-cases', 'max-calls', 'judge-model', 'concurrency', 'max-usd', 'keep-transcripts', 'out', 'trusted-skill'],
|
|
254
|
+
regrade: ['skill', 'answers', 'judge-model', 'samples', 'max-calls', 'max-usd', 'out', 'trusted-skill'],
|
|
255
|
+
};
|
|
256
|
+
function refuseUnknownFlags(command, flags) {
|
|
257
|
+
const unknown = Object.keys(flags).filter((k) => !COMMAND_FLAGS[command].includes(k));
|
|
258
|
+
if (!unknown.length) return;
|
|
259
|
+
console.error(` ✗ REFUSED (${command}): unknown flag ${unknown.map((k) => `--${k}`).join(', ')}; ${command} takes ${COMMAND_FLAGS[command].map((k) => `--${k}`).join(', ')}. Nothing run, no receipt written.`);
|
|
260
|
+
process.exit(2);
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
// Spec 106: the same two commands take one argument. Another positional is refused, exit 2,
|
|
264
|
+
// naming each one given and the one taken, before anything is read. A word after a flag that
|
|
265
|
+
// takes no value (`--trusted-skill maybe`), or a single-dash `-trusted-skill`, used to be
|
|
266
|
+
// dropped with no word.
|
|
267
|
+
function refuseExtraPositionals(command, positional, placeholder) {
|
|
268
|
+
if (positional.length <= 1) return;
|
|
269
|
+
console.error(` ✗ REFUSED (${command}): ${command} takes one argument and was given ${positional.length}: ${positional.join(', ')}. It took ${positional[0]} as ${placeholder}, and would read none of ${positional.slice(1).join(', ')}. A flag that takes no value is given alone. Nothing run, no receipt written.`);
|
|
270
|
+
process.exit(2);
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
// A skill that cannot be loaded because a link stands where it is read (spec 062,
|
|
274
|
+
// register row 5) is refused like any other input, naming the path.
|
|
275
|
+
function loadSkillOrRefuse(skillDir, command) {
|
|
276
|
+
try { return loadSkill(skillDir); } catch (e) {
|
|
277
|
+
if (!e || e.code !== 'SKILL_LINK') throw e;
|
|
278
|
+
console.error(` ✗ REFUSED (${command}): ${e.message}. Nothing measured, no receipt written.`);
|
|
279
|
+
process.exit(2);
|
|
280
|
+
}
|
|
281
|
+
}
|
|
282
|
+
|
|
186
283
|
function usage() {
|
|
187
284
|
console.log(`${PROJECT_NAME} v${RUNNER_VERSION} — continuous verification of agent skills
|
|
188
285
|
|
|
@@ -202,7 +299,7 @@ USAGE
|
|
|
202
299
|
[--summary FILE] [--enforce] [--fail-on-regression true|false]
|
|
203
300
|
${PROJECT_NAME} import <results.json> --from agent-skills-eval|skillgrade [--out DIR]
|
|
204
301
|
${PROJECT_NAME} import <file|dir> --from claude-plugin-eval|skill-creator [--count-errored-runs] [--out DIR]
|
|
205
|
-
${PROJECT_NAME} export <receipt.json> [--to summary-json] [--report-url URL] [--out FILE]
|
|
302
|
+
${PROJECT_NAME} export <receipt.json> [--to summary-json] [--report-url URL] [--out FILE|DIR]
|
|
206
303
|
|
|
207
304
|
ENV
|
|
208
305
|
CLAUDE_PROVIDER api | cli (default: cli — spawns \`claude -p\`, strips ANTHROPIC_API_KEY)
|
|
@@ -273,13 +370,14 @@ NOTES
|
|
|
273
370
|
(driftproof/summary v1) other tools can consume without a receipt parser.`);
|
|
274
371
|
}
|
|
275
372
|
|
|
276
|
-
function slug(s) { return String(s).toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-|-$/g, ''); }
|
|
277
373
|
// An imported receipt whose source records no date carries date_utc null (receipt v0.9).
|
|
278
374
|
function dateStamp(iso) { return iso ? iso.slice(0, 10) : 'undated'; }
|
|
279
375
|
|
|
280
376
|
async function cmdRun(positional, flags) {
|
|
281
377
|
const skillDir = positional[0];
|
|
282
378
|
if (!skillDir) { usage(); process.exit(2); }
|
|
379
|
+
refuseUnknownFlags('run', flags);
|
|
380
|
+
refuseExtraPositionals('run', positional, '<skill-dir>');
|
|
283
381
|
|
|
284
382
|
// Per-project defaults from .driftproofrc (if any); CLI flags override these.
|
|
285
383
|
const rc = loadRc(skillDir);
|
|
@@ -322,7 +420,7 @@ async function cmdRun(positional, flags) {
|
|
|
322
420
|
fs.mkdirSync(outDir, { recursive: true });
|
|
323
421
|
|
|
324
422
|
let skill;
|
|
325
|
-
try { skill =
|
|
423
|
+
try { skill = loadSkillOrRefuse(skillDir, 'run'); } catch (e) {
|
|
326
424
|
// Spec 050 AC-1: a suite whose case ids collide is refused before the projection and
|
|
327
425
|
// before any call, naming the suite file, like the empty suite below.
|
|
328
426
|
if (!e || e.code !== 'DUPLICATE_CASE_ID') throw e;
|
|
@@ -441,10 +539,13 @@ async function cmdRun(positional, flags) {
|
|
|
441
539
|
process.exitCode = 1;
|
|
442
540
|
}
|
|
443
541
|
|
|
444
|
-
|
|
542
|
+
// Spec 062 (register row 3): the name carries the receipt's own hash, so a second
|
|
543
|
+
// run on the same day names a second file, and the file is created, never
|
|
544
|
+
// replaced: a name already taken is a receipt already written.
|
|
545
|
+
const base = receiptBaseName({ skill: skill.name, model: receipt.run.model_id, date: dateStamp(receipt.run.date_utc), hash: receipt.receipt_hash });
|
|
445
546
|
const jsonPath = path.join(outDir, `${base}.json`);
|
|
446
547
|
const mdPath = path.join(outDir, `${base}.summary.md`);
|
|
447
|
-
fs.writeFileSync(jsonPath, JSON.stringify(receipt, null, 2));
|
|
548
|
+
fs.writeFileSync(jsonPath, JSON.stringify(receipt, null, 2), { flag: 'wx' });
|
|
448
549
|
fs.writeFileSync(mdPath, summarizeReceipt(receipt));
|
|
449
550
|
emitted.push(jsonPath);
|
|
450
551
|
|
|
@@ -561,6 +662,8 @@ function cmdStale(positional, flags) {
|
|
|
561
662
|
async function cmdRegrade(positional, flags) {
|
|
562
663
|
const receiptPath = positional[0];
|
|
563
664
|
if (!receiptPath) { console.error(' ✗ regrade: a receipt path is required\n'); usage(); process.exit(2); }
|
|
665
|
+
refuseUnknownFlags('regrade', flags);
|
|
666
|
+
refuseExtraPositionals('regrade', positional, '<receipt.json>');
|
|
564
667
|
for (const f of ['skill', 'answers', 'judge-model']) {
|
|
565
668
|
if (typeof flags[f] !== 'string' || !flags[f]) { console.error(` ✗ REFUSED (regrade): --${f} is required; nothing run, no receipt written.`); process.exit(2); }
|
|
566
669
|
}
|
|
@@ -575,7 +678,7 @@ async function cmdRegrade(positional, flags) {
|
|
|
575
678
|
const maxUsd = checked['max-usd'] !== undefined ? parseFloat(checked['max-usd']) : DEV_MAX_USD;
|
|
576
679
|
const trusted = !!flags['trusted-skill'];
|
|
577
680
|
const outDir = flags.out ? path.resolve(flags.out) : RECEIPTS_DIR;
|
|
578
|
-
const skill =
|
|
681
|
+
const skill = loadSkillOrRefuse(flags.skill, 'regrade');
|
|
579
682
|
const answersFile = path.resolve(flags.answers);
|
|
580
683
|
const answersBytes = fs.readFileSync(answersFile);
|
|
581
684
|
const answers = (JSON.parse(answersBytes.toString('utf8')) || {}).answers || {};
|
|
@@ -618,14 +721,16 @@ async function cmdRegrade(positional, flags) {
|
|
|
618
721
|
if (!valid) { console.error(' ✗ regraded receipt FAILED schema validation:', JSON.stringify(errors, null, 2)); process.exitCode = 1; }
|
|
619
722
|
if (!verifyReceiptHash(out)) { console.error(' ✗ receipt_hash does not verify'); process.exitCode = 1; }
|
|
620
723
|
fs.mkdirSync(outDir, { recursive: true });
|
|
621
|
-
const base =
|
|
724
|
+
const base = receiptBaseName({ skill: out.skill.name, model: out.run.model_id, tag: `regrade-${fileSlug(judgeModel)}`, date: dateStamp(out.run.date_utc), hash: out.receipt_hash });
|
|
622
725
|
const jsonPath = path.join(outDir, `${base}.json`);
|
|
623
|
-
fs.writeFileSync(jsonPath, JSON.stringify(out, null, 2));
|
|
726
|
+
fs.writeFileSync(jsonPath, JSON.stringify(out, null, 2), { flag: 'wx' });
|
|
624
727
|
fs.writeFileSync(path.join(outDir, `${base}.summary.md`), summarizeReceipt(out));
|
|
625
728
|
const provenance = { ...result.provenance, answers: { file: path.basename(answersFile), sha256: sha256(answersBytes), graded: plan.draws } };
|
|
626
|
-
|
|
729
|
+
// The sidecar's name is lib/receipt.js's, which the directory readers skip by (spec 069).
|
|
730
|
+
const sidecar = `${base}${REGRADE_SIDECAR.suffix}`;
|
|
731
|
+
fs.writeFileSync(path.join(outDir, sidecar), JSON.stringify(provenance, null, 2));
|
|
627
732
|
console.log(` → ${path.relative(process.cwd(), jsonPath)} (${result.calls} judge calls)`);
|
|
628
|
-
console.log(` → verification_level ${out.verification_level}; provenance: ${
|
|
733
|
+
console.log(` → verification_level ${out.verification_level}; provenance: ${sidecar}\n`);
|
|
629
734
|
}
|
|
630
735
|
|
|
631
736
|
function cmdValidate(positional) {
|
|
@@ -677,6 +782,7 @@ function cmdBadge(positional, flags) {
|
|
|
677
782
|
const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
678
783
|
refuseUnverified(receipt, p, 'badge');
|
|
679
784
|
refuseAmbiguous(receipt, p, 'badge');
|
|
785
|
+
refuseInvalid(receipt, p, 'badge');
|
|
680
786
|
if (flags.svg) {
|
|
681
787
|
// SPEC 036. The drawn badge: state, model, date, lift and uncertainty, with the
|
|
682
788
|
// machine token in its data attributes. After the same hash check as the JSON.
|
|
@@ -714,7 +820,8 @@ function cmdBadge(positional, flags) {
|
|
|
714
820
|
// read; when --models is given the requested list governs, so a model with no
|
|
715
821
|
// receipt is `refused` and can be the worst state.
|
|
716
822
|
function cmdBadgeSet(dir, flags) {
|
|
717
|
-
const
|
|
823
|
+
const listed = listReceipts(dir);
|
|
824
|
+
const files = listed.map((l) => l.file);
|
|
718
825
|
if (files.length === 0) {
|
|
719
826
|
console.error(` \u2717 REFUSED (badge): ${dir} holds no receipts; a badge over nothing is a badge about nothing.`);
|
|
720
827
|
process.exit(2);
|
|
@@ -722,13 +829,17 @@ function cmdBadgeSet(dir, flags) {
|
|
|
722
829
|
// Every receipt is verified before it is displayed, exactly as the
|
|
723
830
|
// single-receipt path does (spec 026 AC-20): a set is not a way around the
|
|
724
831
|
// check, and the file that fails is named.
|
|
725
|
-
|
|
726
|
-
|
|
727
|
-
|
|
832
|
+
// An unreadable file is left to the decision, which fails its model closed and
|
|
833
|
+
// names it (spec 030 AC-2), as `decide` does.
|
|
834
|
+
for (const { file, receipt } of listed) {
|
|
835
|
+
if (!receipt) continue;
|
|
836
|
+
refuseUnverified(receipt, file, 'badge');
|
|
837
|
+
// Spec 062: validated too. A receipt with duplicate rows is left to the decision's
|
|
838
|
+
// `refused` state (spec 050 AC-3), which fails closed.
|
|
839
|
+
if (!duplicateCaseRows(receipt).length) refuseInvalid(receipt, file, 'badge');
|
|
728
840
|
}
|
|
729
|
-
const models = flags.models ||
|
|
730
|
-
.map((
|
|
731
|
-
.map((r) => r.run && r.run.model_id).filter(Boolean).join(',');
|
|
841
|
+
const models = flags.models || listed
|
|
842
|
+
.map((l) => l.receipt && l.receipt.run && l.receipt.run.model_id).filter(Boolean).join(',');
|
|
732
843
|
const d = decision.decideSet(dir, models);
|
|
733
844
|
if (flags['github-output']) { console.log(decision.githubOutputLines(d)); return; }
|
|
734
845
|
const badge = decision.badgeEndpointForSet(d);
|
|
@@ -781,10 +892,11 @@ function cmdDecide(positional, flags) {
|
|
|
781
892
|
// is unreadable and fails closed on both, naming which (spec 030 AC-2,
|
|
782
893
|
// absence-vs-unreadable). Refusing it here would collapse that distinction
|
|
783
894
|
// into an exit code.
|
|
784
|
-
for (const
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
895
|
+
for (const { file, receipt } of listReceipts(dir)) {
|
|
896
|
+
if (!receipt) continue;
|
|
897
|
+
refuseUnverified(receipt, file, 'decide');
|
|
898
|
+
// Spec 062 (register row 2): and validated, as badge is.
|
|
899
|
+
if (!duplicateCaseRows(receipt).length) refuseInvalid(receipt, file, 'decide');
|
|
788
900
|
}
|
|
789
901
|
const failOnRegression = String(flags['fail-on-regression'] ?? 'true') !== 'false';
|
|
790
902
|
const d = decision.decideSet(dir, flags.models, { failOnRegression });
|
|
@@ -854,8 +966,10 @@ function cmdImport(positional, flags) {
|
|
|
854
966
|
// Spec 049 (approval F-2): the two formats it adds produce several documents per skill, model and
|
|
855
967
|
// day (one per iteration or per results directory), so their receipt name carries the document's
|
|
856
968
|
// own hash and one import never overwrites another's receipt. The same document names the same file.
|
|
857
|
-
|
|
858
|
-
|
|
969
|
+
// The same base-name rule as a run's (spec 062), with the source document's hash in
|
|
970
|
+
// the receipt hash's place, so the name spec 049 gave an import is unchanged.
|
|
971
|
+
const doc = receipt.run.import && ANTHROPIC_TOOLS.includes(from) ? receipt.run.import.source_sha256 : null;
|
|
972
|
+
const base = receiptBaseName({ skill: receipt.skill.name, model: receipt.run.model_id, tag: `imported-${fileSlug(from)}`, date: dateStamp(receipt.run.date_utc), hash: doc });
|
|
859
973
|
const jsonPath = path.join(outDir, `${base}.json`);
|
|
860
974
|
fs.writeFileSync(jsonPath, JSON.stringify(receipt, null, 2));
|
|
861
975
|
if (read !== p) console.log(`read ${path.relative(process.cwd(), read)}`);
|
|
@@ -869,7 +983,7 @@ function cmdImport(positional, flags) {
|
|
|
869
983
|
function cmdExport(positional, flags) {
|
|
870
984
|
const p = positional[0];
|
|
871
985
|
const to = flags.to || 'summary-json';
|
|
872
|
-
if (!p) { console.error('usage: driftproof export <receipt.json> [--to summary-json] [--report-url URL] [--out FILE]'); process.exit(2); }
|
|
986
|
+
if (!p) { console.error('usage: driftproof export <receipt.json> [--to summary-json] [--report-url URL] [--out FILE|DIR]'); process.exit(2); }
|
|
873
987
|
if (to !== 'summary-json') { console.error(`unknown export target "${to}" — supported: summary-json`); process.exit(2); }
|
|
874
988
|
const { toSummaryJson } = require('../lib/export');
|
|
875
989
|
const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
@@ -878,10 +992,14 @@ function cmdExport(positional, flags) {
|
|
|
878
992
|
const summary = toSummaryJson(receipt, { reportUrl: typeof flags['report-url'] === 'string' ? flags['report-url'] : null });
|
|
879
993
|
const json = JSON.stringify(summary, null, 2);
|
|
880
994
|
if (flags.out) {
|
|
881
|
-
|
|
995
|
+
// An existing directory gets the receipt's base name with the summary suffix, the name
|
|
996
|
+
// listReceipts skips beside its receipt (spec 107); any other path is written as given.
|
|
997
|
+
let shown = flags.out;
|
|
998
|
+
if (fs.existsSync(path.resolve(shown)) && fs.statSync(path.resolve(shown)).isDirectory()) shown = path.join(shown, path.basename(p, '.json') + SUMMARY_SIDECAR.suffix);
|
|
999
|
+
const out = path.resolve(shown);
|
|
882
1000
|
fs.mkdirSync(path.dirname(out), { recursive: true });
|
|
883
1001
|
fs.writeFileSync(out, json + '\n');
|
|
884
|
-
console.log(`summary written to ${
|
|
1002
|
+
console.log(`summary written to ${shown} (${summary.verdict}, delta ${summary.delta === null ? 'n/a' : summary.delta})`);
|
|
885
1003
|
} else {
|
|
886
1004
|
console.log(json);
|
|
887
1005
|
}
|
package/config.js
CHANGED
|
@@ -9,7 +9,7 @@ const PROJECT_NAME = 'driftproof';
|
|
|
9
9
|
// Bumped whenever the runner's behaviour or receipt-generation semantics change
|
|
10
10
|
// in a way that could affect results. Recorded into every receipt as
|
|
11
11
|
// run.runner_version so a receipt is reproducible against a known engine.
|
|
12
|
-
const RUNNER_VERSION = '0.
|
|
12
|
+
const RUNNER_VERSION = '0.12.0';
|
|
13
13
|
|
|
14
14
|
// The eval format we CONSUME (we deliberately do not invent our own).
|
|
15
15
|
const SUITE_FORMAT = 'agentskills.io/evals';
|
package/lib/decision.js
CHANGED
|
@@ -17,11 +17,10 @@
|
|
|
17
17
|
// here, in lib/, rather than in the shell, because `bin/driftproof badge` needs
|
|
18
18
|
// it too and because a decision state is a property of a receipt.
|
|
19
19
|
|
|
20
|
-
const fs = require('fs');
|
|
21
20
|
const path = require('path');
|
|
22
21
|
const { EFFECT_FLOOR } = require('../config');
|
|
23
|
-
const { shortModel, githubOutputEntry, receiptVerdict, drawsLine, UNDERPOWERED_LINE } = require('./verdict');
|
|
24
|
-
const { duplicateCaseRows, ambiguityLine } = require('./receipt');
|
|
22
|
+
const { shortModel, githubOutputEntry, receiptVerdict, readCases, drawsLine, UNDERPOWERED_LINE } = require('./verdict');
|
|
23
|
+
const { duplicateCaseRows, ambiguityLine, suiteNarrowed, listReceipts, fileSlug } = require('./receipt');
|
|
25
24
|
|
|
26
25
|
// The six decision states (spec 030 AC-4).
|
|
27
26
|
//
|
|
@@ -96,37 +95,48 @@ function decisionState(receipt) {
|
|
|
96
95
|
const level = receipt.verification_level;
|
|
97
96
|
const kind = receipt.run && receipt.run.answered_by ? receipt.run.answered_by.kind : undefined;
|
|
98
97
|
if (level !== 'TESTED' || kind !== 'model') return 'not measured';
|
|
98
|
+
// Spec 062 (register row 1): THE MEASURED CASES BEFORE THE INCOMPLETE RULE. Until this an
|
|
99
|
+
// incomplete run read `inconclusive` before any case was read, so one unmeasured case hid a
|
|
100
|
+
// regression another case had measured: the same receipt read `regression`, exit 1, with every
|
|
101
|
+
// case measured, and `inconclusive`, exit 0, with one with-skill arm unmeasured (the register's
|
|
102
|
+
// pass 2 probe 7). A case that separated down was measured whatever happened to the others.
|
|
103
|
+
if (readCases(receipt).some((c) => c.state === 'separated-down')) return 'regression';
|
|
99
104
|
const cmp = receipt.comparison || {};
|
|
100
|
-
|
|
105
|
+
// Incomplete: a case failed (run.status), or the receipt ran fewer cases than its suite holds
|
|
106
|
+
// (spec 062, register row 4).
|
|
107
|
+
if ((receipt.run && receipt.run.status === 'incomplete') || suiteNarrowed(receipt) || typeof cmp.delta !== 'number') return 'inconclusive';
|
|
101
108
|
// Spec 035: the measured states are the receipt verdict's, read per case by the
|
|
102
109
|
// band rule (lib/verdict.js receiptVerdict), not the aggregate delta against the floor.
|
|
103
|
-
|
|
104
|
-
//
|
|
105
|
-
|
|
110
|
+
const verdict = receiptVerdict(receipt).verdict;
|
|
111
|
+
// Spec 062: every verdict has a state. NOT_MEASURED is reachable here - a receipt whose every
|
|
112
|
+
// case was dropped reads it by spec 036's no_readable_case rung - and until this it had no row
|
|
113
|
+
// in the table, so the state was undefined and the model left the set (pass 6 V3).
|
|
114
|
+
if (!Object.hasOwn(MEASURED_STATE, verdict)) throw new Error(`decisionState: the verdict ${JSON.stringify(verdict)} has no decision state`);
|
|
115
|
+
return MEASURED_STATE[verdict];
|
|
106
116
|
}
|
|
107
117
|
|
|
108
|
-
const MEASURED_STATE = { REGRESSED: 'regression', PASSED: 'helped', UNDERPOWERED: 'underpowered', NO_EFFECT: 'no detected effect' };
|
|
118
|
+
const MEASURED_STATE = { REGRESSED: 'regression', PASSED: 'helped', UNDERPOWERED: 'underpowered', NO_EFFECT: 'no detected effect', NOT_MEASURED: 'not measured' };
|
|
109
119
|
|
|
110
120
|
// The worst state in a set, by STATE_ORDER. An empty set has no decision, and
|
|
111
|
-
// says so with null rather than defaulting to something benign.
|
|
121
|
+
// says so with null rather than defaulting to something benign. A state this
|
|
122
|
+
// ordering does not know is refused (spec 062): skipping it dropped its model
|
|
123
|
+
// from the decision, which is a pass by omission.
|
|
112
124
|
function worstState(states) {
|
|
113
125
|
let worst = null;
|
|
114
126
|
for (const s of states) {
|
|
115
127
|
const i = STATE_ORDER.indexOf(s);
|
|
116
|
-
if (i < 0)
|
|
128
|
+
if (i < 0) throw new Error(`worstState: unknown decision state ${JSON.stringify(s)}`);
|
|
117
129
|
if (worst === null || i < STATE_ORDER.indexOf(worst)) worst = s;
|
|
118
130
|
}
|
|
119
131
|
return worst;
|
|
120
132
|
}
|
|
121
133
|
|
|
122
|
-
// Every receipt in a directory,
|
|
123
|
-
//
|
|
124
|
-
//
|
|
134
|
+
// Every receipt file in a directory, by lib/receipt.js listReceipts (spec 062): the files that
|
|
135
|
+
// parse and carry receipt_hash or results and are no sidecar by name and shape (spec 069), and the
|
|
136
|
+
// files that do not parse. Until spec 062 it was
|
|
137
|
+
// every `.json` but the summaries and badge.json, so a regrade sidecar was read as a receipt.
|
|
125
138
|
function receiptFiles(dir) {
|
|
126
|
-
return
|
|
127
|
-
.filter((f) => f.endsWith('.json') && !f.includes('.summary.') && f !== 'badge.json')
|
|
128
|
-
.sort()
|
|
129
|
-
.map((f) => path.join(dir, f));
|
|
139
|
+
return listReceipts(dir).map((e) => e.file);
|
|
130
140
|
}
|
|
131
141
|
|
|
132
142
|
// Parse the `models` input the same way the runner does: comma-separated, with
|
|
@@ -159,9 +169,11 @@ function matchesModel(receiptModelId, requested) {
|
|
|
159
169
|
function fileNamesModel(file, requested) {
|
|
160
170
|
const base = path.basename(file);
|
|
161
171
|
// Both spellings, for the reason `matchesModel` takes both: a receipt for
|
|
162
|
-
// `claude-haiku-4-5-20251001` answers a request for `claude-haiku-4-5`.
|
|
172
|
+
// `claude-haiku-4-5-20251001` answers a request for `claude-haiku-4-5`. Each is
|
|
173
|
+
// looked for in the form the runner writes it, lib/receipt.js fileSlug (spec 062): an
|
|
174
|
+
// id with a dot or a capital is not in the file name as itself.
|
|
163
175
|
for (const id of new Set([requested, shortModel(requested)])) {
|
|
164
|
-
if (base.includes(`-${id}-`)) return true;
|
|
176
|
+
if (base.includes(`-${fileSlug(id)}-`)) return true;
|
|
165
177
|
}
|
|
166
178
|
return false;
|
|
167
179
|
}
|
|
@@ -173,12 +185,8 @@ function fileNamesModel(file, requested) {
|
|
|
173
185
|
// row that is simply absent. That is the whole of AC-2: absence is not a pass.
|
|
174
186
|
function decideSet(dir, requestedCsv, { failOnRegression = true } = {}) {
|
|
175
187
|
const requested = parseModels(requestedCsv);
|
|
176
|
-
const
|
|
177
|
-
const
|
|
178
|
-
let receipt = null, error = null;
|
|
179
|
-
try { receipt = JSON.parse(fs.readFileSync(f, 'utf8')); } catch (e) { error = e.message; }
|
|
180
|
-
return { file: f, receipt, error };
|
|
181
|
-
});
|
|
188
|
+
const loaded = listReceipts(dir);
|
|
189
|
+
const files = loaded.map((l) => l.file);
|
|
182
190
|
const used = new Set();
|
|
183
191
|
const rows = requested.map((model) => {
|
|
184
192
|
// Spec 050 AC-6: EVERY readable receipt that matches, not the first. Until this the
|
package/lib/export.js
CHANGED
|
@@ -8,8 +8,9 @@
|
|
|
8
8
|
// Documented in docs/interop.md; snapshot-tested in the gate.
|
|
9
9
|
|
|
10
10
|
const { verdictFromReceipt } = require('./verdict');
|
|
11
|
+
const { SUMMARY_SIDECAR } = require('./receipt');
|
|
11
12
|
|
|
12
|
-
const SUMMARY_FORMAT =
|
|
13
|
+
const SUMMARY_FORMAT = SUMMARY_SIDECAR.format;
|
|
13
14
|
const SUMMARY_FORMAT_VERSION = '1';
|
|
14
15
|
|
|
15
16
|
// Build the summary object for one receipt. Deterministic for a given receipt:
|
package/lib/init.js
CHANGED
|
@@ -11,7 +11,9 @@ const path = require('path');
|
|
|
11
11
|
// <dir>/evals/evals.json — an eval suite with 3 example cases, each rubric
|
|
12
12
|
// anchored at 0.80 (the scoring convention the
|
|
13
13
|
// example suite and Report #001 use)
|
|
14
|
-
// <dir>/.driftproofrc
|
|
14
|
+
// <dir>/.driftproofrc - per-project run defaults (budget, models). No samples,
|
|
15
|
+
// max_cases or judge_model: a skill directory's rc may not
|
|
16
|
+
// set them, and `run` ignores them there (spec 062).
|
|
15
17
|
//
|
|
16
18
|
// NEVER overwrites an existing file — every write is guarded, and an existing
|
|
17
19
|
// path is reported as "skipped". Safe to re-run.
|
|
@@ -89,7 +91,6 @@ function rcTemplate() {
|
|
|
89
91
|
return {
|
|
90
92
|
_comment: 'Per-project driftproof defaults. CLI flags override these. See `driftproof help`.',
|
|
91
93
|
models: 'claude-haiku-4-5',
|
|
92
|
-
samples: 5,
|
|
93
94
|
max_usd: 2,
|
|
94
95
|
};
|
|
95
96
|
}
|
package/lib/receipt.js
CHANGED
|
@@ -57,20 +57,28 @@ const BAND_RULE = 'per arm: the sample standard deviation (n-1) of the per-case
|
|
|
57
57
|
const FAILED_STATUSES = ['failed_timeout', 'failed_unmeasured'];
|
|
58
58
|
function caseFailed(c) { return !!(c && c.case_status && c.case_status !== 'ok'); }
|
|
59
59
|
|
|
60
|
-
|
|
60
|
+
// Spec 062 (register row 2): the table is read by OWN key and the cache is a Map. Until this
|
|
61
|
+
// the lookup was `SCHEMA_FILES[version]`, so a receipt whose schema_version named an
|
|
62
|
+
// Object.prototype property ("constructor", "toString") found something truthy, fell back to
|
|
63
|
+
// the current schema's cached validator and read VALID; and a version nobody knew was checked
|
|
64
|
+
// against the current schema as though it had claimed it. A version that is not a row here has
|
|
65
|
+
// no schema, and a receipt with no schema is not valid.
|
|
66
|
+
const _validators = new Map();
|
|
61
67
|
// Lazily compile the JSON Schema validator (ajv) for a given version. Kept lazy
|
|
62
68
|
// so the library can be required without ajv present (pure hashing utilities).
|
|
69
|
+
// Returns null for a version the table does not hold.
|
|
63
70
|
function getValidator(version) {
|
|
64
|
-
|
|
65
|
-
if (_validators
|
|
71
|
+
if (typeof version !== 'string' || !Object.hasOwn(SCHEMA_FILES, version)) return null;
|
|
72
|
+
if (_validators.has(version)) return _validators.get(version);
|
|
66
73
|
let Ajv;
|
|
67
74
|
// The schema is JSON Schema draft 2020-12, so use ajv's 2020 build.
|
|
68
75
|
try { Ajv = require('ajv/dist/2020'); }
|
|
69
76
|
catch (_e) { throw new Error('receipt validation requires the `ajv` package (npm install)'); }
|
|
70
|
-
const schema = JSON.parse(fs.readFileSync(path.join(__dirname, '..', 'spec', SCHEMA_FILES[
|
|
77
|
+
const schema = JSON.parse(fs.readFileSync(path.join(__dirname, '..', 'spec', SCHEMA_FILES[version]), 'utf8'));
|
|
71
78
|
const ajv = new Ajv({ allErrors: true, strict: false });
|
|
72
|
-
|
|
73
|
-
|
|
79
|
+
const validate = ajv.compile(schema);
|
|
80
|
+
_validators.set(version, validate);
|
|
81
|
+
return validate;
|
|
74
82
|
}
|
|
75
83
|
|
|
76
84
|
// Compute the receipt_hash: sha256 over the canonical receipt JSON with the
|
|
@@ -118,9 +126,16 @@ function ambiguityLine(dups) {
|
|
|
118
126
|
|
|
119
127
|
// Validate against the schema matching the receipt's own schema_version (so both
|
|
120
128
|
// v0.1 and v0.2 receipts validate). Returns { valid, errors, version }.
|
|
129
|
+
//
|
|
130
|
+
// Spec 062: a schema_version that is absent, or is not a version the table holds, is invalid,
|
|
131
|
+
// with no fallback to the current schema. Absent is not read as current: every receipt the
|
|
132
|
+
// runner and the importers write carries one.
|
|
121
133
|
function validateReceipt(receipt) {
|
|
122
|
-
const version =
|
|
134
|
+
const version = receipt && typeof receipt === 'object' ? receipt.schema_version : undefined;
|
|
123
135
|
const validate = getValidator(version);
|
|
136
|
+
if (!validate) {
|
|
137
|
+
return { valid: false, errors: [{ instancePath: '/schema_version', message: `unknown schema_version ${JSON.stringify(version === undefined ? null : version)}: not a version this validator holds (${Object.keys(SCHEMA_FILES).join(', ')}); no other schema was tried` }], version };
|
|
138
|
+
}
|
|
124
139
|
const valid = validate(receipt);
|
|
125
140
|
const errors = valid ? [] : (validate.errors || []);
|
|
126
141
|
// v0.8 (spec 043 AC-2): run.judge.samples and run.counts.judge_samples_per_generation
|
|
@@ -137,6 +152,94 @@ function validateReceipt(receipt) {
|
|
|
137
152
|
return { valid: errors.length === 0, errors, version };
|
|
138
153
|
}
|
|
139
154
|
|
|
155
|
+
// Spec 062 (register row 4): whether a receipt ran fewer cases than its suite holds. The receipt
|
|
156
|
+
// records the ids it ran (results.cases) and the suite's count (suite.case_count), not the
|
|
157
|
+
// suite's ids, so a receipt narrows its suite when either arm names fewer distinct case ids than
|
|
158
|
+
// the suite counts (R-5). A failed case keeps its rows, so it is not a cut. A
|
|
159
|
+
// narrowed receipt is not a reading of the suite its suite_hash names, and both readers treat it
|
|
160
|
+
// as incomplete (lib/verdict.js receiptVerdict, lib/decision.js decisionState).
|
|
161
|
+
function suiteNarrowed(receipt) {
|
|
162
|
+
const n = receipt && receipt.suite ? receipt.suite.case_count : undefined;
|
|
163
|
+
if (!Number.isInteger(n)) return false;
|
|
164
|
+
const ids = { with_skill: new Set(), baseline: new Set() };
|
|
165
|
+
for (const c of (receipt.results && receipt.results.cases) || []) if (c && ids[c.mode]) ids[c.mode].add(c.id);
|
|
166
|
+
return ids.with_skill.size < n || ids.baseline.size < n;
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
// ── receipt files (spec 062, register row 3) ────────────────────────────────
|
|
170
|
+
// The writer's naming rule and the readers' listing rule, in one module. Until this the name was
|
|
171
|
+
// assembled in bin/driftproof and read back by lib/decision.js with a different rule: a second run
|
|
172
|
+
// on the same day overwrote the first, a regrade sidecar in the directory was read as a receipt
|
|
173
|
+
// that did not verify, and a model id with a dot was looked for in a name that held its slug.
|
|
174
|
+
|
|
175
|
+
// The file-name form of a skill name or a model id. Named for what it is rather than `slug`: the
|
|
176
|
+
// probe-copies rule reads an exported name, and spec 054's probe has a heading-anchor `slug` of its
|
|
177
|
+
// own that is not this function (spec 062 A-062-1).
|
|
178
|
+
function fileSlug(s) { return String(s).toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-|-$/g, ''); }
|
|
179
|
+
|
|
180
|
+
// <skill>-<model>[-<tag>]-<date>[-<hash12>]. `hash` is the first 12 characters of the receipt's
|
|
181
|
+
// own receipt_hash for a run, so two runs on one day name two files; an import passes its source
|
|
182
|
+
// document's hash, which keeps the name spec 049 gave it.
|
|
183
|
+
function receiptBaseName({ skill, model, date, hash = null, tag = null }) {
|
|
184
|
+
return [fileSlug(skill), fileSlug(model), ...(tag ? [tag] : []), date, ...(hash ? [String(hash).slice(0, 12)] : [])].join('-');
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
// The regrade provenance sidecar: its name is the receipt's base name with this suffix, and its
|
|
188
|
+
// `format` names this version. `regrade`'s writers read both from here (lib/regrade.js the format,
|
|
189
|
+
// bin/driftproof the name), so the listing rule below is the writer's own (spec 069).
|
|
190
|
+
const REGRADE_SIDECAR = { suffix: '.regrade.json', format: 'driftproof-regrade/2' };
|
|
191
|
+
const REGRADE_FAMILY = REGRADE_SIDECAR.format.split('/')[0];
|
|
192
|
+
|
|
193
|
+
// The summary export kept beside its receipt: `export --out <dir>` names it the receipt's base name
|
|
194
|
+
// with this suffix, and lib/export.js stamps this format on it (spec 107). It carries the receipt's
|
|
195
|
+
// receipt_hash as a reference, so the listing rule below skips it by name and shape.
|
|
196
|
+
const SUMMARY_SIDECAR = { suffix: '.summary.json', format: 'driftproof/summary' };
|
|
197
|
+
|
|
198
|
+
// The files a writer puts beside a receipt that carry its receipt_hash and are not receipts. A file
|
|
199
|
+
// is one of these only when its name AND its shape match (spec 069, the operator's ruling Q1): a
|
|
200
|
+
// name alone or a shape alone is not enough, so a receipt renamed, or a sidecar renamed, is not
|
|
201
|
+
// passed over. None of them carries `results`.
|
|
202
|
+
// - regrade provenance: `format` is a version of the regrade format. The published Report 011
|
|
203
|
+
// sidecars carry the version before the writer's current one, and read as sidecars.
|
|
204
|
+
// - surface record (spec 042's run): `receipt` names the receipt it sits beside, which is its own
|
|
205
|
+
// name with the suffix replaced by `.json`.
|
|
206
|
+
// - summary export (spec 107): `format` is the summary format and `format_version` a string of
|
|
207
|
+
// digits.
|
|
208
|
+
const RECEIPT_SIDECARS = [
|
|
209
|
+
{ kind: 'regrade provenance', suffix: REGRADE_SIDECAR.suffix,
|
|
210
|
+
shape: (doc) => typeof doc.format === 'string' && new RegExp(`^${REGRADE_FAMILY}/[0-9]+$`).test(doc.format) },
|
|
211
|
+
{ kind: 'surface record', suffix: '.surface.json',
|
|
212
|
+
shape: (doc, name) => doc.receipt === `${name.slice(0, -'.surface.json'.length)}.json` },
|
|
213
|
+
{ kind: 'summary export', suffix: SUMMARY_SIDECAR.suffix,
|
|
214
|
+
shape: (doc) => doc.format === SUMMARY_SIDECAR.format && typeof doc.format_version === 'string' && /^[0-9]+$/.test(doc.format_version) },
|
|
215
|
+
];
|
|
216
|
+
function sidecarKind(name, doc) {
|
|
217
|
+
if (Object.hasOwn(doc, 'results')) return null;
|
|
218
|
+
const s = RECEIPT_SIDECARS.find((x) => name.endsWith(x.suffix) && name.length > x.suffix.length && x.shape(doc, name));
|
|
219
|
+
return s ? s.kind : null;
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
// Every receipt in a directory, sorted by file name: { file, receipt, error }. A `.json` file that
|
|
223
|
+
// parses to an object carrying `receipt_hash` or `results` is returned, unless it is a sidecar by
|
|
224
|
+
// name and shape (above), so the caller's hash check refuses by name a receipt with either field
|
|
225
|
+
// removed, rather than reading it as absent (spec 062 A-062-3 for `receipt_hash`, spec 069 for
|
|
226
|
+
// `results`). An object carrying neither is not a receipt and is not returned: a badge, a stale
|
|
227
|
+
// document. A `.json` file that does not parse IS returned, with receipt null and the
|
|
228
|
+
// parser's message: it may be a receipt, and spec 030 AC-2 fails its model closed as unreadable.
|
|
229
|
+
function listReceipts(dir) {
|
|
230
|
+
const out = [];
|
|
231
|
+
for (const name of fs.readdirSync(dir).filter((f) => f.endsWith('.json')).sort()) {
|
|
232
|
+
const file = path.join(dir, name);
|
|
233
|
+
let receipt;
|
|
234
|
+
try { receipt = JSON.parse(fs.readFileSync(file, 'utf8')); } catch (e) { out.push({ file, receipt: null, error: e.message }); continue; }
|
|
235
|
+
if (!receipt || typeof receipt !== 'object' || Array.isArray(receipt)) continue;
|
|
236
|
+
if (!Object.hasOwn(receipt, 'results') && !Object.hasOwn(receipt, 'receipt_hash')) continue;
|
|
237
|
+
if (sidecarKind(name, receipt)) continue;
|
|
238
|
+
out.push({ file, receipt, error: null });
|
|
239
|
+
}
|
|
240
|
+
return out;
|
|
241
|
+
}
|
|
242
|
+
|
|
140
243
|
function mean(nums) {
|
|
141
244
|
if (!nums.length) return 0;
|
|
142
245
|
return round(nums.reduce((a, b) => a + b, 0) / nums.length);
|
|
@@ -339,5 +442,5 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
|
|
|
339
442
|
|
|
340
443
|
module.exports = {
|
|
341
444
|
buildReceipt, sealReceipt, computeReceiptHash, verifyReceiptHash, validateReceipt, BAND_RULE, comparisonOf, caseFailed, FAILED_STATUSES,
|
|
342
|
-
duplicateCaseRows, ambiguityLine,
|
|
445
|
+
duplicateCaseRows, ambiguityLine, suiteNarrowed, fileSlug, receiptBaseName, listReceipts, REGRADE_SIDECAR, SUMMARY_SIDECAR,
|
|
343
446
|
};
|
package/lib/regrade.js
CHANGED
|
@@ -25,7 +25,7 @@
|
|
|
25
25
|
const { sha256 } = require('./canonical');
|
|
26
26
|
const { judgeCase, attest, outcomeFor, resolveCallTimeoutMs } = require('./run');
|
|
27
27
|
const { judgeSettings, promptTemplateHash, rubricHash } = require('./judge');
|
|
28
|
-
const { buildReceipt, verifyReceiptHash, FAILED_STATUSES } = require('./receipt');
|
|
28
|
+
const { buildReceipt, verifyReceiptHash, FAILED_STATUSES, REGRADE_SIDECAR } = require('./receipt');
|
|
29
29
|
const { acrossDraws } = require('./sampling');
|
|
30
30
|
const { resolveModel, surfaceForModel, isMeteredSurface } = require('./provider');
|
|
31
31
|
const { priceForModel, assertRegistered } = require('./models');
|
|
@@ -250,7 +250,7 @@ async function regradeReceipt({ receipt, skill, answers, judgeModel, samples, op
|
|
|
250
250
|
// revision and the archived arms are in the receipt (spec 043 AC-4) and are not repeated
|
|
251
251
|
// here; this sidecar is bound to the receipt by its hash.
|
|
252
252
|
const provenance = {
|
|
253
|
-
format:
|
|
253
|
+
format: REGRADE_SIDECAR.format,
|
|
254
254
|
receipt_hash: out.receipt_hash,
|
|
255
255
|
regraded_from: { receipt_hash: receipt.receipt_hash, runner_version: receipt.run.runner_version, date_utc: receipt.run.date_utc, judge: receipt.run.judge },
|
|
256
256
|
generated_at_basis: receipt.run.generated_at || (receipt.run.arms && Object.values(receipt.run.arms).some((x) => x && x.generated_at))
|
package/lib/skill.js
CHANGED
|
@@ -30,23 +30,57 @@ function pastBound(bound, limit, value, what) {
|
|
|
30
30
|
const IGNORE_DIRS = new Set(['.git', 'node_modules', 'evals']);
|
|
31
31
|
const IGNORE_FILES = new Set(['.DS_Store']);
|
|
32
32
|
|
|
33
|
+
// Spec 062 (register row 5): A LINK IS REFUSED WHERE THE SKILL IS READ. Until this SKILL.md was
|
|
34
|
+
// read through a link, so a skill directory could make any file the operator can read into the
|
|
35
|
+
// system prompt, while the walk skipped links and content_hash covered none of the bytes used
|
|
36
|
+
// (the SHA-256 of empty input, unchanged when the target changed). A link is neither followed nor
|
|
37
|
+
// skipped: the skill is refused, naming it. The skill directory itself may be reached by a link;
|
|
38
|
+
// what is under it may not.
|
|
39
|
+
function linkRefusal(rel, base) {
|
|
40
|
+
const e = new Error(`${rel} under ${base} is a symbolic link; a skill is read only from its own directory, so a link is refused rather than followed or skipped`);
|
|
41
|
+
e.code = 'SKILL_LINK'; e.path = rel;
|
|
42
|
+
return e;
|
|
43
|
+
}
|
|
44
|
+
// Every component from the skill directory down to `rel` is a plain entry, not a link. An absent
|
|
45
|
+
// component is left to the caller, whose own message names what is missing.
|
|
46
|
+
function assertNoLink(dir, rel) {
|
|
47
|
+
let cur = dir;
|
|
48
|
+
for (const part of rel.split('/')) {
|
|
49
|
+
cur = path.join(cur, part);
|
|
50
|
+
let st;
|
|
51
|
+
try { st = fs.lstatSync(cur); } catch (_e) { return; }
|
|
52
|
+
if (st.isSymbolicLink()) throw linkRefusal(path.relative(dir, cur), dir);
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
|
|
33
56
|
// Bounded (spec 026 AC-13): the walk stops at SKILL_MAX_DEPTH directories
|
|
34
57
|
// below the skill dir, SKILL_MAX_FILES bundled files, and SKILL_MAX_BYTES of
|
|
35
58
|
// them together, and throws naming the bound the moment one is passed, so a
|
|
36
59
|
// pathological tree is refused before its bytes are read into memory.
|
|
37
|
-
|
|
60
|
+
//
|
|
61
|
+
// `read` maps a path relative to the skill dir to bytes the caller has already read, so the hash
|
|
62
|
+
// covers those bytes and not a second read of the same path (spec 062: SKILL.md).
|
|
63
|
+
function walkFiles(dir, base = dir, acc = [], state = { bytes: 0 }, depth = 0, read = {}) {
|
|
38
64
|
if (depth > SKILL_MAX_DEPTH) throw pastBound('SKILL_MAX_DEPTH', SKILL_MAX_DEPTH, depth, `directory depth under ${base}`);
|
|
39
65
|
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
|
66
|
+
if (entry.isSymbolicLink()) {
|
|
67
|
+
// A link under an ignored name is never read, so it is left as the name is; `evals` is
|
|
68
|
+
// checked where the suite is read.
|
|
69
|
+
if (IGNORE_DIRS.has(entry.name) || IGNORE_FILES.has(entry.name)) continue;
|
|
70
|
+
throw linkRefusal(path.relative(base, path.join(dir, entry.name)), base);
|
|
71
|
+
}
|
|
40
72
|
if (entry.isDirectory()) {
|
|
41
73
|
if (IGNORE_DIRS.has(entry.name)) continue;
|
|
42
|
-
walkFiles(path.join(dir, entry.name), base, acc, state, depth + 1);
|
|
74
|
+
walkFiles(path.join(dir, entry.name), base, acc, state, depth + 1, read);
|
|
43
75
|
} else if (entry.isFile() && !IGNORE_FILES.has(entry.name)) {
|
|
44
76
|
const abs = path.join(dir, entry.name);
|
|
77
|
+
const rel = path.relative(base, abs);
|
|
45
78
|
if (acc.length + 1 > SKILL_MAX_FILES) throw pastBound('SKILL_MAX_FILES', SKILL_MAX_FILES, acc.length + 1, `bundled files under ${base}`);
|
|
46
|
-
const
|
|
79
|
+
const given = Object.hasOwn(read, rel) ? read[rel] : null;
|
|
80
|
+
const size = given ? given.length : fs.statSync(abs).size;
|
|
47
81
|
state.bytes += size;
|
|
48
82
|
if (state.bytes > SKILL_MAX_BYTES) throw pastBound('SKILL_MAX_BYTES', SKILL_MAX_BYTES, state.bytes, `bundled bytes under ${base}`);
|
|
49
|
-
acc.push({ path:
|
|
83
|
+
acc.push({ path: rel, bytes: given || fs.readFileSync(abs) });
|
|
50
84
|
}
|
|
51
85
|
}
|
|
52
86
|
return acc;
|
|
@@ -55,14 +89,17 @@ function walkFiles(dir, base = dir, acc = [], state = { bytes: 0 }, depth = 0) {
|
|
|
55
89
|
function loadSkill(skillDir) {
|
|
56
90
|
const dir = path.resolve(skillDir);
|
|
57
91
|
const skillMdPath = path.join(dir, 'SKILL.md');
|
|
92
|
+
assertNoLink(dir, 'SKILL.md');
|
|
58
93
|
if (!fs.existsSync(skillMdPath)) {
|
|
59
94
|
throw new Error(`no SKILL.md found in ${dir}`);
|
|
60
95
|
}
|
|
61
|
-
const
|
|
96
|
+
const skillMdBytes = fs.readFileSync(skillMdPath);
|
|
97
|
+
const skillMd = skillMdBytes.toString('utf8');
|
|
62
98
|
|
|
63
99
|
// Content hash over SKILL.md + every bundled file (evals/ excluded — the suite
|
|
64
100
|
// is hashed separately so a suite edit doesn't masquerade as a skill change).
|
|
65
|
-
|
|
101
|
+
// SKILL.md enters it as the bytes read above, the bytes the run is given.
|
|
102
|
+
const files = walkFiles(dir, dir, [], { bytes: 0 }, 0, { 'SKILL.md': skillMdBytes });
|
|
66
103
|
const contentHash = sha256Files(files);
|
|
67
104
|
|
|
68
105
|
// Parse skill name/version from front-matter or the first H1; fall back to dir.
|
|
@@ -72,6 +109,7 @@ function loadSkill(skillDir) {
|
|
|
72
109
|
|
|
73
110
|
// Load the eval suite.
|
|
74
111
|
const suitePath = path.join(dir, 'evals', 'evals.json');
|
|
112
|
+
assertNoLink(dir, 'evals/evals.json');
|
|
75
113
|
if (!fs.existsSync(suitePath)) {
|
|
76
114
|
throw new Error(`no evals/evals.json found in ${dir}`);
|
|
77
115
|
}
|
package/lib/verdict.js
CHANGED
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
const crypto = require('crypto');
|
|
5
5
|
const { EFFECT_FLOOR, POWER_Z } = require('../config');
|
|
6
6
|
const { bandOf } = require('./reuse');
|
|
7
|
-
const { duplicateCaseRows, ambiguityLine } = require('./receipt');
|
|
7
|
+
const { duplicateCaseRows, ambiguityLine, suiteNarrowed } = require('./receipt');
|
|
8
8
|
|
|
9
9
|
// Spec 050 AC-3. A receipt with two rows for one (id, mode) has no verdict: which row
|
|
10
10
|
// the byId map below kept would decide it. The error carries a code so a caller that
|
|
@@ -186,7 +186,11 @@ function receiptVerdict(receipt) {
|
|
|
186
186
|
// incomplete: a case had an arm that could not be measured) is not verdicted
|
|
187
187
|
// either. spec/RECEIPT.md has said since v0.3.1 that a drift report must not
|
|
188
188
|
// compute a verdict from one; the badge is the same reader with a shorter path.
|
|
189
|
-
|
|
189
|
+
//
|
|
190
|
+
// Spec 062 (register row 4): a receipt that ran fewer cases than its suite holds is incomplete
|
|
191
|
+
// too, by the same route. Its suite_hash names the whole suite, so a verdict read off it would
|
|
192
|
+
// be a verdict on cases that were never run.
|
|
193
|
+
const incomplete = !!(receipt && receipt.run && receipt.run.status === 'incomplete') || suiteNarrowed(receipt);
|
|
190
194
|
// Spec 036 A-036-4: WHICH REFUSAL FIRED. The guard below is one line and says only that
|
|
191
195
|
// some refusal did; a surface that renders NOT_MEASURED in words owes its reader the one
|
|
192
196
|
// that actually fired, and cannot get it from a boolean. `notMeasuredRoutes` names the same
|
|
@@ -201,20 +205,7 @@ function receiptVerdict(receipt) {
|
|
|
201
205
|
// the rung edges. A guard and a route list that disagree read RED there.
|
|
202
206
|
const routes = notMeasuredRoutes(level, kind, cmp.delta, incomplete);
|
|
203
207
|
if (level !== 'TESTED' || kind !== 'model' || typeof cmp.delta !== 'number' || incomplete) return { verdict: 'NOT_MEASURED', cases: [], drawsNeeded: null, notMeasured: routes };
|
|
204
|
-
const
|
|
205
|
-
for (const c of (receipt.results && receipt.results.cases) || []) {
|
|
206
|
-
if (c.case_status && c.case_status !== 'ok') continue;
|
|
207
|
-
const e = byId.get(c.id) || {};
|
|
208
|
-
e[c.mode] = c;
|
|
209
|
-
byId.set(c.id, e);
|
|
210
|
-
}
|
|
211
|
-
const cases = [];
|
|
212
|
-
for (const [id, e] of byId) {
|
|
213
|
-
const w = armOf(e.with_skill);
|
|
214
|
-
const b = armOf(e.baseline);
|
|
215
|
-
if (!w || !b) continue;
|
|
216
|
-
cases.push({ id, ...caseRule(b, w) });
|
|
217
|
-
}
|
|
208
|
+
const cases = readCases(receipt);
|
|
218
209
|
// Spec 036 A-036-3, THE RUNG R-5 NEVER HAD. R-5 reads its ladder off the cases, and
|
|
219
210
|
// every rung below REGRESSED is a statement about what the cases showed. NO_EFFECT is
|
|
220
211
|
// the bottom rung and it is a MEASUREMENT: "no separation detected at this sample size".
|
|
@@ -231,6 +222,31 @@ function receiptVerdict(receipt) {
|
|
|
231
222
|
return { verdict, cases, drawsNeeded: verdict === 'UNDERPOWERED' ? receiptDrawsNeeded(cases) : null, notMeasured: null };
|
|
232
223
|
}
|
|
233
224
|
|
|
225
|
+
// R-1 to R-4 over every readable case of a receipt, with none of R-5's refusals applied: a case
|
|
226
|
+
// with a failed arm, or an arm with no readable band, is left out. receiptVerdict reads its ladder
|
|
227
|
+
// off this list; lib/decision.js reads it BEFORE the incomplete rule (spec 062, register row 1),
|
|
228
|
+
// because a case that was measured and separated down is a measured regression however many other
|
|
229
|
+
// cases went unmeasured.
|
|
230
|
+
function readCases(receipt) {
|
|
231
|
+
const dups = duplicateCaseRows(receipt);
|
|
232
|
+
if (dups.length) throw new AmbiguousReceiptError(dups);
|
|
233
|
+
const byId = new Map();
|
|
234
|
+
for (const c of (receipt && receipt.results && receipt.results.cases) || []) {
|
|
235
|
+
if (c.case_status && c.case_status !== 'ok') continue;
|
|
236
|
+
const e = byId.get(c.id) || {};
|
|
237
|
+
e[c.mode] = c;
|
|
238
|
+
byId.set(c.id, e);
|
|
239
|
+
}
|
|
240
|
+
const cases = [];
|
|
241
|
+
for (const [id, e] of byId) {
|
|
242
|
+
const w = armOf(e.with_skill);
|
|
243
|
+
const b = armOf(e.baseline);
|
|
244
|
+
if (!w || !b) continue;
|
|
245
|
+
cases.push({ id, ...caseRule(b, w) });
|
|
246
|
+
}
|
|
247
|
+
return cases;
|
|
248
|
+
}
|
|
249
|
+
|
|
234
250
|
// Derive the verdict object from a receipt. Returns
|
|
235
251
|
// { verdict, delta, floor, model, message, color, drawsNeeded }
|
|
236
252
|
function verdictFromReceipt(receipt) {
|
|
@@ -300,6 +316,6 @@ function githubOutputLines(receipt) {
|
|
|
300
316
|
|
|
301
317
|
module.exports = {
|
|
302
318
|
verdictFromReceipt, badgeEndpoint, githubOutputLines, githubOutputEntry, shortModel, VERDICTS,
|
|
303
|
-
receiptVerdict, caseRule, armOf, drawsLine, receiptDrawsTaken, UNDERPOWERED_LINE, NOT_MEASURED_ROUTES,
|
|
319
|
+
receiptVerdict, readCases, caseRule, armOf, drawsLine, receiptDrawsTaken, UNDERPOWERED_LINE, NOT_MEASURED_ROUTES,
|
|
304
320
|
AmbiguousReceiptError,
|
|
305
321
|
};
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "driftproof",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.12.0",
|
|
4
4
|
"description": "A dated proof that this skill, this hash, this model, still helps: run a skill's eval suite with and without the skill across model versions, emit hash-verified dated receipts, and diff receipts into drift reports.",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
6
|
"keywords": [
|