driftproof 0.11.1 → 0.11.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +49 -14
- package/bin/driftproof +194 -8
- package/config/models.json +3 -2
- package/config.js +2 -2
- package/lib/counts.js +104 -0
- package/lib/decision.js +40 -7
- package/lib/diff.js +12 -3
- package/lib/export.js +4 -1
- package/lib/importers-anthropic.js +451 -0
- package/lib/importers.js +50 -13
- package/lib/receipt.js +72 -3
- package/lib/regrade.js +266 -0
- package/lib/reuse.js +144 -9
- package/lib/run.js +39 -4
- package/lib/skill.js +26 -9
- package/lib/stale.js +194 -0
- package/lib/verdict.js +18 -0
- package/package.json +2 -2
- package/spec/RECEIPT.md +103 -13
- package/spec/receipt.schema.json +330 -15
- package/spec/receipt.v0.7.schema.json +1563 -0
- package/spec/receipt.v0.8.schema.json +1677 -0
- package/spec/stale.v1.schema.json +353 -0
package/README.md
CHANGED
|
@@ -13,17 +13,19 @@ there were too few draws to tell, the receipt says so.
|
|
|
13
13
|
— live badge for the bundled `commit-message-conventions` example, generated from its own receipt.
|
|
14
14
|
|
|
15
15
|
**[Quickstart](https://driftproofhq.com/#quickstart)** ·
|
|
16
|
-
**[Latest report](https://driftproofhq.com/reports/
|
|
16
|
+
**[Latest report](https://driftproofhq.com/reports/011/)** ·
|
|
17
17
|
[driftproofhq.com](https://driftproofhq.com)
|
|
18
18
|
|
|
19
19
|
Driftproof **consumes** the [`agentskills.io/evals`](https://agentskills.io) eval
|
|
20
20
|
format; it does not invent its own.
|
|
21
21
|
|
|
22
|
-
📊 **
|
|
23
|
-
hand-entered
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
22
|
+
📊 **Eleven published reports** (each re-derived from committed files, nothing
|
|
23
|
+
hand-entered: Driftproof's receipts, or for Report #010 the upstream harness's own
|
|
24
|
+
output), spanning seven published report types, the newest being instrument
|
|
25
|
+
comparison. The ten that measure with Driftproof read its own arms by one
|
|
26
|
+
band-based, floor-gated verdict rule and differ in what moves underneath the
|
|
27
|
+
skill — or, in the value report, in which axes are measured; or, in the instrument re-measurement, in the
|
|
28
|
+
instrument itself; or, in the instrument comparison, in which instrument measures:
|
|
27
29
|
|
|
28
30
|
- **[Report #001](https://driftproofhq.com/reports/001/)** — *release drift*: ten
|
|
29
31
|
public agent skills across a current-vs-previous Sonnet release; 9 of 10 showed
|
|
@@ -76,11 +78,35 @@ axes are measured; or, in the instrument re-measurement, in the instrument itsel
|
|
|
76
78
|
receipts, which is what makes the delta attributable to the model rather than
|
|
77
79
|
to the instrument. One case sits inside the verdict on the effect floor alone
|
|
78
80
|
and the report names it.
|
|
81
|
+
- **[Report #009](https://driftproofhq.com/reports/009/)** — *instrument
|
|
82
|
+
comparison*: three skills from one plugin, one case each, measured by Claude
|
|
83
|
+
Code's native plugin eval and by Driftproof on the same SKILL.md bytes, task
|
|
84
|
+
prompts and rubrics. The two tools apply different treatments and grade
|
|
85
|
+
differently, so the report reads them side by side, ranks neither, and states
|
|
86
|
+
what each can and cannot establish. Both tools judged with `claude-opus-5`,
|
|
87
|
+
which departs from the judge policy, so its figures are not comparable with
|
|
88
|
+
Reports #001 to #008.
|
|
89
|
+
- **[Report #010](https://driftproofhq.com/reports/010/)** — *instrument
|
|
90
|
+
comparison*: one skill's own behavioural eval from its upstream repository, run
|
|
91
|
+
five times under each of two Claude Code configurations at one commit: 8 passed
|
|
92
|
+
and 2 failed, both failures on one expectation the skill's ADR instructions do
|
|
93
|
+
not state. No Driftproof measurement was taken and no Driftproof judge ran, and
|
|
94
|
+
the report does not isolate what caused the variation.
|
|
95
|
+
- **[Report #011](https://driftproofhq.com/reports/011/)** — *release drift*:
|
|
96
|
+
Claude Opus 5.5 on its release day, measured on Report #009's three skills and
|
|
97
|
+
cases, one case each, beside a fresh Claude Opus 5 arm on the same Claude Code
|
|
98
|
+
version. Read by the runner's own comparison of the with-skill arms, the three
|
|
99
|
+
skills read 1 with no separation detected and 2 with not enough draws to conclude
|
|
100
|
+
at the effect floor, which is not evidence that nothing changed. A second table
|
|
101
|
+
sets Report #009's own Claude Opus 5 receipts beside this run's; between them the
|
|
102
|
+
Claude Code version, its host build and the date all changed. Every draw was
|
|
103
|
+
judged with `claude-opus-5`, which departs from the judge policy, so its figures
|
|
104
|
+
are not comparable with Reports #001 to #008.
|
|
79
105
|
|
|
80
106
|
✍️ The launch essay, **[Three model releases later: what actually happens to agent
|
|
81
|
-
skills](https://driftproofhq.com/writing/three-releases/)**, reads all
|
|
107
|
+
skills](https://driftproofhq.com/writing/three-releases/)**, reads all eleven reports
|
|
82
108
|
together: what moves underneath a skill, what the skill costs to run, and what a
|
|
83
|
-
corrected instrument did to three published results. Revised 2026-09-
|
|
109
|
+
corrected instrument did to three published results. Revised 2026-09-23; every
|
|
84
110
|
figure in it is gate-checked against the report page it cites.
|
|
85
111
|
|
|
86
112
|
## Why
|
|
@@ -213,6 +239,7 @@ npx driftproof badge receipt.json --out badges/my-skill.json
|
|
|
213
239
|
# epistemics — no fabricated hashes, excluded from drift verdicts), and emit
|
|
214
240
|
# the minimal stable summary other tools can consume. See docs/interop.md.
|
|
215
241
|
npx driftproof import results.json --from agent-skills-eval # or: skillgrade
|
|
242
|
+
npx driftproof import evals/results/<timestamp>/ --from claude-plugin-eval # or: skill-creator
|
|
216
243
|
npx driftproof export receipt.json --to summary-json
|
|
217
244
|
```
|
|
218
245
|
|
|
@@ -296,7 +323,7 @@ A receipt is the unit of evidence — one JSON document conforming to
|
|
|
296
323
|
|
|
297
324
|
```jsonc
|
|
298
325
|
{
|
|
299
|
-
"schema_version": "0.
|
|
326
|
+
"schema_version": "0.9",
|
|
300
327
|
"skill": { "name": "commit-message-conventions", "version": "0.2.0",
|
|
301
328
|
"content_hash": "…sha256 over SKILL.md + bundled files…" },
|
|
302
329
|
"suite": { "format": "agentskills.io/evals", "suite_hash": "…", "case_count": 10 },
|
|
@@ -305,7 +332,7 @@ A receipt is the unit of evidence — one JSON document conforming to
|
|
|
305
332
|
"model_release_date": "2025-10-01",
|
|
306
333
|
"provider": "anthropic",
|
|
307
334
|
"surface": "claude-cli",
|
|
308
|
-
"runner_version": "0.11.
|
|
335
|
+
"runner_version": "0.11.3",
|
|
309
336
|
"date_utc": "2026-07-27T…Z",
|
|
310
337
|
"registry": "registered",
|
|
311
338
|
"transcripts": "hashes-only",
|
|
@@ -384,7 +411,7 @@ jobs:
|
|
|
384
411
|
runs-on: ubuntu-latest
|
|
385
412
|
steps:
|
|
386
413
|
- uses: actions/checkout@v4
|
|
387
|
-
- uses: driftproofhq/driftproof@v0.11.
|
|
414
|
+
- uses: driftproofhq/driftproof@v0.11.3
|
|
388
415
|
with:
|
|
389
416
|
skill-dir: skills/my-skill
|
|
390
417
|
models: claude-haiku-4-5
|
|
@@ -432,9 +459,15 @@ the skill hurt — but they **never render as success** on any of the three
|
|
|
432
459
|
surfaces: not in the badge, not in the summary row, and not in the check title,
|
|
433
460
|
which carries a `::warning` naming the state and the models it came from.
|
|
434
461
|
|
|
462
|
+
A receipt set that does not say one thing is `REFUSED` as well. Two receipts for
|
|
463
|
+
one requested model, or a receipt with two rows for one case and arm, fail the job
|
|
464
|
+
naming the files or the case, because a decision that depends on which one was read
|
|
465
|
+
is not a pass. Each invocation of the Action writes to its own directory and uploads
|
|
466
|
+
its own artifact, so a job may run it more than once, one skill per step.
|
|
467
|
+
|
|
435
468
|
Step outputs: `verdict`, `delta` (of the model the worst decision came from),
|
|
436
|
-
`worst_state`, `regressed_models`, `missing_models
|
|
437
|
-
described in [`action.yml`](action.yml).
|
|
469
|
+
`worst_state`, `regressed_models`, `missing_models`, `receipts_dir` and
|
|
470
|
+
`artifact_name`. Each is described in [`action.yml`](action.yml).
|
|
438
471
|
|
|
439
472
|
For a free CI dry-run with **zero model calls**, set `DRIFTPROOF_STUB=1` in the job
|
|
440
473
|
env — the runner returns canned receipts so the wiring can be tested without spend
|
|
@@ -479,7 +512,7 @@ site, so it reflects a real dated run, not a hand-set color.
|
|
|
479
512
|
|
|
480
513
|
## Reports
|
|
481
514
|
|
|
482
|
-
|
|
515
|
+
Eleven reports are published, spanning seven report types. A report page lives at a
|
|
483
516
|
draft path — `docs/reports/NNN-draft/` — until the publish sequence renames it, and
|
|
484
517
|
`scripts/build-public.sh` excludes every `*-draft/` path from the published tree
|
|
485
518
|
(see the roll at the top of this README, and
|
|
@@ -487,6 +520,8 @@ draft path — `docs/reports/NNN-draft/` — until the publish sequence renames
|
|
|
487
520
|
Each report and every verdict in it are **re-derived from the receipts** committed
|
|
488
521
|
under [`receipts/`](receipts/)
|
|
489
522
|
(`receipts/report-001/` … `receipts/report-006/`) — nothing is hand-entered.
|
|
523
|
+
Report #010 takes no Driftproof measurement and has no receipts: it is re-derived
|
|
524
|
+
from the upstream harness's output files, published beside its page.
|
|
490
525
|
|
|
491
526
|
Driftproof does **not** commit third-party skill content. Each `SKILL.md` is
|
|
492
527
|
fetched at run time from a pinned commit and verified by sha256 against
|
package/bin/driftproof
CHANGED
|
@@ -8,7 +8,7 @@ const { PROJECT_NAME, RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES, DEV_MAX_USD, DEV_MA
|
|
|
8
8
|
const { loadSkill } = require('../lib/skill');
|
|
9
9
|
const { runSkillOnModel, summarizeReceipt, projectCalls, answeredLine, band, uncertaintyStr } = require('../lib/run');
|
|
10
10
|
const { SAMPLING } = require('../lib/sampling');
|
|
11
|
-
const { validateReceipt, verifyReceiptHash } = require('../lib/receipt');
|
|
11
|
+
const { validateReceipt, verifyReceiptHash, duplicateCaseRows, ambiguityLine } = require('../lib/receipt');
|
|
12
12
|
const { buildDriftReport, revisionPairProblem } = require('../lib/diff');
|
|
13
13
|
const { surfaceForModel, isSubscriptionSurface, resolveModel, evalUser } = require('../lib/provider');
|
|
14
14
|
const { estimateRunCostUSD, BudgetTracker } = require('../lib/cost');
|
|
@@ -16,6 +16,8 @@ const { registryStatus, assertRegistered, REGISTRY_PATH } = require('../lib/mode
|
|
|
16
16
|
const { verdictFromReceipt, badgeEndpoint, githubOutputLines, drawsLine, UNDERPOWERED_LINE } = require('../lib/verdict');
|
|
17
17
|
const decision = require('../lib/decision');
|
|
18
18
|
const { scaffoldInit } = require('../lib/init');
|
|
19
|
+
const { planRegrade, regradeReceipt } = require('../lib/regrade');
|
|
20
|
+
const { sha256 } = require('../lib/canonical');
|
|
19
21
|
|
|
20
22
|
// Output dirs default to the USER's current directory, not the package dir, so a
|
|
21
23
|
// global/npx install writes receipts/transcripts into the project being tested
|
|
@@ -59,6 +61,20 @@ function refuseUnverified(receipt, file, command) {
|
|
|
59
61
|
process.exit(4);
|
|
60
62
|
}
|
|
61
63
|
|
|
64
|
+
// ── AN AMBIGUOUS RECEIPT IS NOT RENDERED (spec 050 AC-3) ─────────────────────
|
|
65
|
+
// A receipt with two rows for one (id, mode) verifies and, before spec 050, validated:
|
|
66
|
+
// the loader let two cases share an id, and the verdict kept whichever row came last.
|
|
67
|
+
// The surfaces that render ONE receipt refuse it here, after the hash check, the way
|
|
68
|
+
// they refuse a tampered one: nothing on stdout, no file, the pair on stderr, exit 2.
|
|
69
|
+
// The set surfaces (decide, badge <dir>) do not come here: they decide the model
|
|
70
|
+
// `refused`, which fails the job closed and keeps the row that says why.
|
|
71
|
+
function refuseAmbiguous(receipt, file, command) {
|
|
72
|
+
const dups = duplicateCaseRows(receipt);
|
|
73
|
+
if (!dups.length) return;
|
|
74
|
+
console.error(` ✗ REFUSED (${command}): ${path.basename(file)} is ambiguous: ${ambiguityLine(dups)}. A verdict over it would depend on which row was read; nothing rendered.`);
|
|
75
|
+
process.exit(2);
|
|
76
|
+
}
|
|
77
|
+
|
|
62
78
|
// Read an optional .driftproofrc (JSON) for per-project run defaults. Looked up
|
|
63
79
|
// in the skill dir first (where `driftproof init` writes it), then the CWD. CLI
|
|
64
80
|
// flags always win over the rc; the rc wins over built-in defaults.
|
|
@@ -75,7 +91,7 @@ function loadRc(skillDir) {
|
|
|
75
91
|
|
|
76
92
|
// Flags that never take a value, so `--trusted-skill <skill-dir>` keeps the dir
|
|
77
93
|
// positional instead of swallowing it as the flag's value.
|
|
78
|
-
const BOOLEAN_FLAGS = new Set(['trusted-skill', 'svg']);
|
|
94
|
+
const BOOLEAN_FLAGS = new Set(['trusted-skill', 'svg', 'count-errored-runs', 'strict', 'no-harness-check']);
|
|
79
95
|
|
|
80
96
|
// ── THE INPUT CONTRACT (spec 026 AC-9, F5) ────────────────────────────────────
|
|
81
97
|
// A COPY of action/lib.sh's RE_* rules, in the shape spec 028's fixture holds
|
|
@@ -176,11 +192,16 @@ USAGE
|
|
|
176
192
|
[--judge-model M] [--concurrency N] [--max-usd N]
|
|
177
193
|
[--keep-transcripts] [--out DIR] [--trusted-skill]
|
|
178
194
|
${PROJECT_NAME} diff <receiptA.json> <receiptB.json> [--out FILE] [--mode release|revision]
|
|
195
|
+
${PROJECT_NAME} regrade <receipt.json> --skill <dir> --answers <file> --judge-model <id>
|
|
196
|
+
[--samples N] [--max-calls N] [--max-usd N] [--out DIR] [--trusted-skill]
|
|
197
|
+
${PROJECT_NAME} stale <receipt.json>... [--skill DIR] [--suite FILE] [--model ID] [--judge ID]
|
|
198
|
+
[--harness-version V | --no-harness-check] [--strict] [--json [path]]
|
|
179
199
|
${PROJECT_NAME} validate <receipt.json>
|
|
180
200
|
${PROJECT_NAME} badge <receipt.json|receipts-dir> [--models a,b] [--out FILE] [--github-output] [--svg [--href URL]]
|
|
181
201
|
${PROJECT_NAME} decide <receipts-dir> --models a,b [--github-output] [--badge FILE]
|
|
182
202
|
[--summary FILE] [--enforce] [--fail-on-regression true|false]
|
|
183
203
|
${PROJECT_NAME} import <results.json> --from agent-skills-eval|skillgrade [--out DIR]
|
|
204
|
+
${PROJECT_NAME} import <file|dir> --from claude-plugin-eval|skill-creator [--count-errored-runs] [--out DIR]
|
|
184
205
|
${PROJECT_NAME} export <receipt.json> [--to summary-json] [--report-url URL] [--out FILE]
|
|
185
206
|
|
|
186
207
|
ENV
|
|
@@ -228,12 +249,33 @@ NOTES
|
|
|
228
249
|
epistemics: verification_level DECLARED (never TESTED), surface "external",
|
|
229
250
|
source "imported/<tool>", and NO fabricated hashes. Imported receipts are
|
|
230
251
|
excluded from drift verdicts (diff reports NOT MEASURED). See docs/interop.md.
|
|
252
|
+
- import --from claude-plugin-eval reads \`claude plugin eval\`'s aggregate-result.json
|
|
253
|
+
(or its --json output); --from skill-creator reads a benchmark.json (skill-creator's,
|
|
254
|
+
or skill-up's, with its result.json beside it). Given a directory, it finds the one
|
|
255
|
+
document and refuses none or several. Runs the source recorded an error for are kept
|
|
256
|
+
out of the band and listed (--count-errored-runs puts them back); anything the source
|
|
257
|
+
does not record is written as unknown and named in run.import.notices.
|
|
258
|
+
- stale compares what each receipt recorded (model, harness, skill, suite, judge,
|
|
259
|
+
template, each case's rubric) with what would run today, read from the flags,
|
|
260
|
+
then .driftproofrc; a value nobody gives is unknown, never current. It says per
|
|
261
|
+
arm whether the result stands, needs a rerun, or only a regrade, and prints the
|
|
262
|
+
command to run next. No model call, no network. Exit 0 current, 1 stale, 3
|
|
263
|
+
something unknown or a newer model, 2 an error; --strict turns 3 into 1.
|
|
264
|
+
- regrade grades a receipt's saved answers again with another judge (or the
|
|
265
|
+
same judge under a new template): what lib/reuse.js's triage calls
|
|
266
|
+
"regrade". --answers is {"answers": {"<generation_hash>": "<text>"}}; every
|
|
267
|
+
answer is checked against its hash before any call. The generation side of
|
|
268
|
+
the new receipt is the original's, draw for draw; --samples defaults to the
|
|
269
|
+
original's. The receipt records the re-judge (its archived arms, judged_at and
|
|
270
|
+
grader_revision); <receipt>.regrade.json beside it names the original receipt
|
|
271
|
+
and the answers file.
|
|
231
272
|
- export --to summary-json emits the minimal stable interchange summary
|
|
232
273
|
(driftproof/summary v1) other tools can consume without a receipt parser.`);
|
|
233
274
|
}
|
|
234
275
|
|
|
235
276
|
function slug(s) { return String(s).toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-|-$/g, ''); }
|
|
236
|
-
|
|
277
|
+
// An imported receipt whose source records no date carries date_utc null (receipt v0.9).
|
|
278
|
+
function dateStamp(iso) { return iso ? iso.slice(0, 10) : 'undated'; }
|
|
237
279
|
|
|
238
280
|
async function cmdRun(positional, flags) {
|
|
239
281
|
const skillDir = positional[0];
|
|
@@ -279,7 +321,14 @@ async function cmdRun(positional, flags) {
|
|
|
279
321
|
const outDir = flags.out ? path.resolve(flags.out) : RECEIPTS_DIR;
|
|
280
322
|
fs.mkdirSync(outDir, { recursive: true });
|
|
281
323
|
|
|
282
|
-
|
|
324
|
+
let skill;
|
|
325
|
+
try { skill = loadSkill(skillDir); } catch (e) {
|
|
326
|
+
// Spec 050 AC-1: a suite whose case ids collide is refused before the projection and
|
|
327
|
+
// before any call, naming the suite file, like the empty suite below.
|
|
328
|
+
if (!e || e.code !== 'DUPLICATE_CASE_ID') throw e;
|
|
329
|
+
console.error(` ✗ REFUSED: the suite at ${path.join(path.resolve(skillDir), 'evals', 'evals.json')}: ${e.message}. Nothing measured, no receipt written.`);
|
|
330
|
+
process.exit(2);
|
|
331
|
+
}
|
|
283
332
|
const nCases = maxCases ? Math.min(maxCases, skill.suite.caseCount) : skill.suite.caseCount;
|
|
284
333
|
// Spec 026 AC-5 (F2): a suite with no cases measures nothing, and a receipt
|
|
285
334
|
// over nothing is a receipt about nothing. Refused here, before the
|
|
@@ -467,6 +516,118 @@ function cmdDiff(positional, flags) {
|
|
|
467
516
|
}
|
|
468
517
|
}
|
|
469
518
|
|
|
519
|
+
// ── stale (spec 053) ────────────────────────────────────────────────────────
|
|
520
|
+
// Whether each receipt's conclusion still stands under what would run today. Exit 0 current, 1 stale,
|
|
521
|
+
// 3 nothing stale but something unknown or a newer model, 2 an unreadable or invalid receipt or a bad
|
|
522
|
+
// flag; --strict turns 3 into 1. `--json` prints the driftproof.stale/1 document; `--json <path>`
|
|
523
|
+
// writes it there, over an earlier stale document, and refuses (exit 2) a path that holds anything
|
|
524
|
+
// else, so `stale --json receipts/*.json` cannot overwrite a receipt (A-053-1, approval F-1).
|
|
525
|
+
function cmdStale(positional, flags) {
|
|
526
|
+
const { staleReport, renderText } = require('../lib/stale');
|
|
527
|
+
const VALUE = ['skill', 'suite', 'model', 'judge', 'harness-version'];
|
|
528
|
+
const BOOL = ['strict', 'no-harness-check'];
|
|
529
|
+
const bad = (msg) => { console.error(` ✗ stale: ${msg}\n`); console.error(` usage: ${PROJECT_NAME} stale <receipt.json>... [--skill DIR] [--suite FILE] [--model ID] [--judge ID] [--harness-version V | --no-harness-check] [--strict] [--json [path]]`); process.exitCode = 2; };
|
|
530
|
+
for (const k of Object.keys(flags)) if (![...VALUE, ...BOOL, 'json'].includes(k)) return bad(`unknown flag --${k}`);
|
|
531
|
+
for (const k of VALUE) if (flags[k] === true) return bad(`--${k} needs a value`);
|
|
532
|
+
if (flags['harness-version'] && flags['no-harness-check']) return bad('--harness-version and --no-harness-check together');
|
|
533
|
+
const files = [...positional];
|
|
534
|
+
let jsonOut = null;
|
|
535
|
+
if (typeof flags.json === 'string') {
|
|
536
|
+
if (fs.existsSync(flags.json)) {
|
|
537
|
+
let earlier = null; try { earlier = JSON.parse(fs.readFileSync(flags.json, 'utf8')); } catch { /* not a stale document */ }
|
|
538
|
+
if (!earlier || earlier.schema !== 'driftproof.stale/1') return bad(`--json ${flags.json} names an existing file that is not a stale document; nothing is written over it (put --json last with a new path, or give --json alone)`);
|
|
539
|
+
}
|
|
540
|
+
jsonOut = flags.json;
|
|
541
|
+
}
|
|
542
|
+
if (!files.length) return bad('no receipt given');
|
|
543
|
+
if (flags.skill) { try { loadSkill(flags.skill); } catch (e) { return bad(`--skill ${flags.skill}: ${e.message}`); } }
|
|
544
|
+
if (flags.suite) { try { JSON.parse(fs.readFileSync(flags.suite, 'utf8')); } catch (e) { return bad(`--suite ${flags.suite}: ${e.message}`); } }
|
|
545
|
+
const doc = staleReport(files, {
|
|
546
|
+
skill: flags.skill || null, suite: flags.suite || null, model: flags.model || null, judge: flags.judge || null,
|
|
547
|
+
harnessVersion: flags['harness-version'] || null, noHarnessCheck: !!flags['no-harness-check'], strict: !!flags.strict,
|
|
548
|
+
rc: loadRc(flags.skill || null),
|
|
549
|
+
});
|
|
550
|
+
if (flags.json) {
|
|
551
|
+
const text = JSON.stringify(doc, null, 2) + '\n';
|
|
552
|
+
if (jsonOut) { fs.writeFileSync(jsonOut, text); process.stdout.write(renderText(doc)); } else process.stdout.write(text);
|
|
553
|
+
} else process.stdout.write(renderText(doc));
|
|
554
|
+
process.exitCode = doc.exit_code;
|
|
555
|
+
}
|
|
556
|
+
|
|
557
|
+
// ── regrade (spec 044) ──────────────────────────────────────────────────────
|
|
558
|
+
// The executor behind lib/reuse.js's `regrade` decision. Every input is checked
|
|
559
|
+
// before the projection, and the projection before any call: the receipt's seal,
|
|
560
|
+
// the skill and suite it was run on, and every answer against its hash (AC-4).
|
|
561
|
+
async function cmdRegrade(positional, flags) {
|
|
562
|
+
const receiptPath = positional[0];
|
|
563
|
+
if (!receiptPath) { console.error(' ✗ regrade: a receipt path is required\n'); usage(); process.exit(2); }
|
|
564
|
+
for (const f of ['skill', 'answers', 'judge-model']) {
|
|
565
|
+
if (typeof flags[f] !== 'string' || !flags[f]) { console.error(` ✗ REFUSED (regrade): --${f} is required; nothing run, no receipt written.`); process.exit(2); }
|
|
566
|
+
}
|
|
567
|
+
const checked = checkNumericInputs(flags, {});
|
|
568
|
+
const receipt = JSON.parse(fs.readFileSync(receiptPath, 'utf8'));
|
|
569
|
+
const judgeModel = resolveModel(flags['judge-model']);
|
|
570
|
+
checkRegistered([receipt.run.model_id], () => judgeModel);
|
|
571
|
+
const origSamples = receipt.run && receipt.run.judge && receipt.run.judge.samples;
|
|
572
|
+
const samples = checked.samples !== undefined ? parseInt(checked.samples, 10) : origSamples;
|
|
573
|
+
if (!Number.isInteger(samples) || samples < 1) { console.error(' ✗ REFUSED (regrade): the receipt records no judge sample count; pass --samples.'); process.exit(2); }
|
|
574
|
+
const maxCalls = checked['max-calls'] !== undefined ? parseInt(checked['max-calls'], 10) : DEV_MAX_CALLS;
|
|
575
|
+
const maxUsd = checked['max-usd'] !== undefined ? parseFloat(checked['max-usd']) : DEV_MAX_USD;
|
|
576
|
+
const trusted = !!flags['trusted-skill'];
|
|
577
|
+
const outDir = flags.out ? path.resolve(flags.out) : RECEIPTS_DIR;
|
|
578
|
+
const skill = loadSkill(flags.skill);
|
|
579
|
+
const answersFile = path.resolve(flags.answers);
|
|
580
|
+
const answersBytes = fs.readFileSync(answersFile);
|
|
581
|
+
const answers = (JSON.parse(answersBytes.toString('utf8')) || {}).answers || {};
|
|
582
|
+
|
|
583
|
+
const plan = planRegrade({ receipt, skill, answers, judgeModel, samples });
|
|
584
|
+
console.log(`\n${PROJECT_NAME} regrade: ${path.basename(receiptPath)}`);
|
|
585
|
+
if (process.env.DRIFTPROOF_STUB === '1') console.log(' STUB RUN (DRIFTPROOF_STUB=1): nothing will answer the judge; the receipt will be UNVERIFIED and grade nothing');
|
|
586
|
+
console.log(` model: ${receipt.run.model_id} (generations kept) judge: ${receipt.run.judge.model_id} → ${judgeModel} samples/answer: ${samples}`);
|
|
587
|
+
if (plan.problems.length) {
|
|
588
|
+
console.error(`\n ✗ REFUSED (regrade): ${plan.problems.length} problem(s) with the inputs; nothing run, no receipt written.`);
|
|
589
|
+
for (const p of plan.problems.slice(0, 20)) console.error(` - ${p}`);
|
|
590
|
+
process.exit(2);
|
|
591
|
+
}
|
|
592
|
+
console.log(` answers to grade: ${plan.draws} projected calls: ${plan.calls} cap: ${maxCalls}`);
|
|
593
|
+
console.log(` projected cost: ~$${plan.usd.toFixed(2)} (the runner's per-call judge estimate; budget $${maxUsd.toFixed(2)}, hard-stop $${(maxUsd * 1.25).toFixed(2)})`);
|
|
594
|
+
if (plan.calls > maxCalls) {
|
|
595
|
+
console.error(`\n ✗ ABORT (cost guard): projected ${plan.calls} calls exceeds --max-calls ${maxCalls}.`);
|
|
596
|
+
process.exit(3);
|
|
597
|
+
}
|
|
598
|
+
if (plan.usd > maxUsd) {
|
|
599
|
+
console.error(`\n ✗ ABORT (cost guard): projected cost ~$${plan.usd.toFixed(2)} exceeds --max-usd $${maxUsd.toFixed(2)}.`);
|
|
600
|
+
process.exit(3);
|
|
601
|
+
}
|
|
602
|
+
let result;
|
|
603
|
+
try {
|
|
604
|
+
result = await regradeReceipt({
|
|
605
|
+
receipt, skill, answers, judgeModel, samples,
|
|
606
|
+
opts: {
|
|
607
|
+
trusted, budget: new BudgetTracker(maxUsd),
|
|
608
|
+
onProgress: (p) => { if (p.phase === 'done') console.log(` ${p.case} / ${p.mode}: ${p.outcome} (${p.score.toFixed(2)} ± ${(p.stddev || 0).toFixed(2)})`); },
|
|
609
|
+
},
|
|
610
|
+
});
|
|
611
|
+
} catch (e) {
|
|
612
|
+
if (e && e.code === 'SUBSTRATE_MISMATCH') { console.error(`\n ✗ REFUSED (substrate mismatch): ${e.message}`); process.exit(5); }
|
|
613
|
+
if (e && e.code === 'BUDGET_HARDSTOP') { console.error(`\n ✗ ABORT (budget hard-stop): ${e.message}`); process.exit(3); }
|
|
614
|
+
throw e;
|
|
615
|
+
}
|
|
616
|
+
const out = result.receipt;
|
|
617
|
+
const { valid, errors } = validateReceipt(out);
|
|
618
|
+
if (!valid) { console.error(' ✗ regraded receipt FAILED schema validation:', JSON.stringify(errors, null, 2)); process.exitCode = 1; }
|
|
619
|
+
if (!verifyReceiptHash(out)) { console.error(' ✗ receipt_hash does not verify'); process.exitCode = 1; }
|
|
620
|
+
fs.mkdirSync(outDir, { recursive: true });
|
|
621
|
+
const base = `${slug(out.skill.name)}-${slug(out.run.model_id)}-regrade-${slug(judgeModel)}-${dateStamp(out.run.date_utc)}`;
|
|
622
|
+
const jsonPath = path.join(outDir, `${base}.json`);
|
|
623
|
+
fs.writeFileSync(jsonPath, JSON.stringify(out, null, 2));
|
|
624
|
+
fs.writeFileSync(path.join(outDir, `${base}.summary.md`), summarizeReceipt(out));
|
|
625
|
+
const provenance = { ...result.provenance, answers: { file: path.basename(answersFile), sha256: sha256(answersBytes), graded: plan.draws } };
|
|
626
|
+
fs.writeFileSync(path.join(outDir, `${base}.regrade.json`), JSON.stringify(provenance, null, 2));
|
|
627
|
+
console.log(` → ${path.relative(process.cwd(), jsonPath)} (${result.calls} judge calls)`);
|
|
628
|
+
console.log(` → verification_level ${out.verification_level}; provenance: ${base}.regrade.json\n`);
|
|
629
|
+
}
|
|
630
|
+
|
|
470
631
|
function cmdValidate(positional) {
|
|
471
632
|
const p = positional[0];
|
|
472
633
|
if (!p) { usage(); process.exit(2); }
|
|
@@ -515,6 +676,7 @@ function cmdBadge(positional, flags) {
|
|
|
515
676
|
if (fs.existsSync(p) && fs.statSync(p).isDirectory()) return cmdBadgeSet(p, flags);
|
|
516
677
|
const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
517
678
|
refuseUnverified(receipt, p, 'badge');
|
|
679
|
+
refuseAmbiguous(receipt, p, 'badge');
|
|
518
680
|
if (flags.svg) {
|
|
519
681
|
// SPEC 036. The drawn badge: state, model, date, lift and uncertainty, with the
|
|
520
682
|
// machine token in its data attributes. After the same hash check as the JSON.
|
|
@@ -661,12 +823,27 @@ function cmdImport(positional, flags) {
|
|
|
661
823
|
const p = positional[0];
|
|
662
824
|
const from = flags.from;
|
|
663
825
|
const { IMPORT_TOOLS, importResults } = require('../lib/importers');
|
|
826
|
+
const { ANTHROPIC_TOOLS, importAnthropic, ImportRefused } = require('../lib/importers-anthropic');
|
|
827
|
+
const tools = [...IMPORT_TOOLS, ...ANTHROPIC_TOOLS];
|
|
664
828
|
if (!p || typeof from !== 'string') {
|
|
665
|
-
console.error(`usage: driftproof import <results.json> --from ${
|
|
829
|
+
console.error(`usage: driftproof import <results.json|dir> --from ${tools.join('|')} [--count-errored-runs] [--out DIR]`);
|
|
666
830
|
process.exit(2);
|
|
667
831
|
}
|
|
668
|
-
|
|
669
|
-
|
|
832
|
+
let receipt;
|
|
833
|
+
let read = p;
|
|
834
|
+
if (ANTHROPIC_TOOLS.includes(from)) {
|
|
835
|
+
// Spec 049: a file or a directory; the clock is read once, into run.import.imported_at.
|
|
836
|
+
try {
|
|
837
|
+
({ receipt, file: read } = importAnthropic(p, { from, importedAt: new Date().toISOString(), countErroredRuns: flags['count-errored-runs'] === true }));
|
|
838
|
+
} catch (e) {
|
|
839
|
+
if (!(e instanceof ImportRefused)) throw e;
|
|
840
|
+
console.error(`✗ import refused, nothing written: ${e.message}`);
|
|
841
|
+
process.exit(1);
|
|
842
|
+
}
|
|
843
|
+
} else {
|
|
844
|
+
const data = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
845
|
+
receipt = importResults(data, { from, importedAt: new Date().toISOString() });
|
|
846
|
+
}
|
|
670
847
|
const { valid, errors } = validateReceipt(receipt);
|
|
671
848
|
if (!valid || !verifyReceiptHash(receipt)) {
|
|
672
849
|
console.error('✗ converted receipt failed validation — nothing written:', JSON.stringify(errors, null, 2));
|
|
@@ -674,12 +851,18 @@ function cmdImport(positional, flags) {
|
|
|
674
851
|
}
|
|
675
852
|
const outDir = flags.out ? path.resolve(flags.out) : RECEIPTS_DIR;
|
|
676
853
|
fs.mkdirSync(outDir, { recursive: true });
|
|
677
|
-
|
|
854
|
+
// Spec 049 (approval F-2): the two formats it adds produce several documents per skill, model and
|
|
855
|
+
// day (one per iteration or per results directory), so their receipt name carries the document's
|
|
856
|
+
// own hash and one import never overwrites another's receipt. The same document names the same file.
|
|
857
|
+
const doc = receipt.run.import && ANTHROPIC_TOOLS.includes(from) ? `-${receipt.run.import.source_sha256.slice(0, 12)}` : '';
|
|
858
|
+
const base = `${slug(receipt.skill.name)}-${slug(receipt.run.model_id)}-imported-${slug(from)}-${dateStamp(receipt.run.date_utc)}${doc}`;
|
|
678
859
|
const jsonPath = path.join(outDir, `${base}.json`);
|
|
679
860
|
fs.writeFileSync(jsonPath, JSON.stringify(receipt, null, 2));
|
|
861
|
+
if (read !== p) console.log(`read ${path.relative(process.cwd(), read)}`);
|
|
680
862
|
console.log(`imported → ${path.relative(process.cwd(), jsonPath)}`);
|
|
681
863
|
console.log(` verification_level: DECLARED (source ${receipt.run.source}) — the source tool's declaration, converted faithfully;`);
|
|
682
864
|
console.log(` no generation hashes (never fabricated); excluded from drift verdicts unless re-run TESTED. See docs/interop.md.`);
|
|
865
|
+
for (const n of (receipt.run.import && receipt.run.import.notices) || []) console.log(` notice: ${n}`);
|
|
683
866
|
}
|
|
684
867
|
|
|
685
868
|
// Emit the minimal stable interchange summary for one receipt (interop, Phase 7).
|
|
@@ -691,6 +874,7 @@ function cmdExport(positional, flags) {
|
|
|
691
874
|
const { toSummaryJson } = require('../lib/export');
|
|
692
875
|
const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
693
876
|
refuseUnverified(receipt, p, 'export');
|
|
877
|
+
refuseAmbiguous(receipt, p, 'export');
|
|
694
878
|
const summary = toSummaryJson(receipt, { reportUrl: typeof flags['report-url'] === 'string' ? flags['report-url'] : null });
|
|
695
879
|
const json = JSON.stringify(summary, null, 2);
|
|
696
880
|
if (flags.out) {
|
|
@@ -710,6 +894,8 @@ async function main() {
|
|
|
710
894
|
case 'init': return cmdInit(positional);
|
|
711
895
|
case 'run': return cmdRun(positional, flags);
|
|
712
896
|
case 'diff': return cmdDiff(positional, flags);
|
|
897
|
+
case 'regrade': return cmdRegrade(positional, flags);
|
|
898
|
+
case 'stale': return cmdStale(positional, flags);
|
|
713
899
|
case 'validate': return cmdValidate(positional);
|
|
714
900
|
case 'badge': return cmdBadge(positional, flags);
|
|
715
901
|
case 'decide': return cmdDecide(positional, flags);
|
package/config/models.json
CHANGED
|
@@ -10,8 +10,9 @@
|
|
|
10
10
|
"models": [
|
|
11
11
|
{ "id": "claude-fable-5-1", "family": "fable", "provider": "anthropic", "released": "2026-09-01", "input_price": 10.0, "output_price": 50.0, "cache_read_price": 0.25, "tier": "frontier", "judge_eligible": false, "surface_verified": "claude-cli", "surface_verified_at": "2026-09-01" },
|
|
12
12
|
{ "id": "claude-fable-5", "family": "fable", "provider": "anthropic", "released": "2026-06-09", "input_price": 10.0, "output_price": 50.0, "tier": "frontier", "judge_eligible": false, "lifecycle": "legacy" },
|
|
13
|
-
{ "id": "claude-opus-5",
|
|
14
|
-
{ "id": "claude-opus-
|
|
13
|
+
{ "id": "claude-opus-5-5", "family": "opus", "provider": "anthropic", "released": "2026-09-22", "input_price": 4.0, "output_price": 20.0, "cache_read_price": 0.2, "tier": "frontier", "judge_eligible": false },
|
|
14
|
+
{ "id": "claude-opus-5", "family": "opus", "provider": "anthropic", "released": "2026-07-24", "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
|
|
15
|
+
{ "id": "claude-opus-4-8", "family": "opus", "provider": "anthropic", "released": "2026-05-28", "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
|
|
15
16
|
{ "id": "claude-opus-4-7", "family": "opus", "provider": "anthropic", "released": null, "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
|
|
16
17
|
{ "id": "claude-opus-4-6", "family": "opus", "provider": "anthropic", "released": null, "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
|
|
17
18
|
{ "id": "claude-opus-4-5-20251101", "family": "opus", "provider": "anthropic", "released": "2025-11-01", "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
|
package/config.js
CHANGED
|
@@ -9,7 +9,7 @@ const PROJECT_NAME = 'driftproof';
|
|
|
9
9
|
// Bumped whenever the runner's behaviour or receipt-generation semantics change
|
|
10
10
|
// in a way that could affect results. Recorded into every receipt as
|
|
11
11
|
// run.runner_version so a receipt is reproducible against a known engine.
|
|
12
|
-
const RUNNER_VERSION = '0.11.
|
|
12
|
+
const RUNNER_VERSION = '0.11.3';
|
|
13
13
|
|
|
14
14
|
// The eval format we CONSUME (we deliberately do not invent our own).
|
|
15
15
|
const SUITE_FORMAT = 'agentskills.io/evals';
|
|
@@ -38,7 +38,7 @@ const SUITE_FORMAT = 'agentskills.io/evals';
|
|
|
38
38
|
// results.aggregates.band_rule, and bands that are null where the formula
|
|
39
39
|
// cannot form. NOT additive for the validator: a v0.5 receipt restamped 0.6
|
|
40
40
|
// is refused. v0.5 is frozen as receipt.v0.5.schema.json.
|
|
41
|
-
const RECEIPT_SCHEMA_VERSION = '0.
|
|
41
|
+
const RECEIPT_SCHEMA_VERSION = '0.9';
|
|
42
42
|
|
|
43
43
|
// Input bounds on the skill loader and the post-checks (spec 026 AC-13, audit
|
|
44
44
|
// A3/A7). lib/skill.js loaded every bundled file with no count, depth, byte,
|
package/lib/counts.js
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// Receipt counts (spec 043, receipt spec v0.8). Two counts describe how a receipt's data
|
|
5
|
+
// came to be, and they are kept apart because they measure different things (Report 006:
|
|
6
|
+
// generation spread ran 3 to 7 times judge spread on the same cases):
|
|
7
|
+
//
|
|
8
|
+
// generations_per_arm independent candidate generations
|
|
9
|
+
// judge_samples_per_generation judge samples taken of each generation
|
|
10
|
+
//
|
|
11
|
+
// A count sits at the narrowest scope that is honest about it: run.counts when it is
|
|
12
|
+
// the same for every case of every arm generated in this run, run.arms.<mode>.counts
|
|
13
|
+
// when it holds for one arm, results.cases[i].counts otherwise. An absent count is
|
|
14
|
+
// unknown, never 1 (agentskills discussion #544, comment 18409751, rule 1).
|
|
15
|
+
|
|
16
|
+
const KINDS = ['generations_per_arm', 'judge_samples_per_generation'];
|
|
17
|
+
|
|
18
|
+
// An arm that carries its own generated_at was generated in an earlier run. run.counts
|
|
19
|
+
// describes this run's generations, so an archived arm never inherits it.
|
|
20
|
+
function archived(receipt, mode) {
|
|
21
|
+
const a = receipt.run && receipt.run.arms && receipt.run.arms[mode];
|
|
22
|
+
return !!(a && a.generated_at);
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
const has = (o, k) => !!o && Object.prototype.hasOwnProperty.call(o, k) && Number.isInteger(o[k]);
|
|
26
|
+
|
|
27
|
+
// Where a reader looks, narrowest first.
|
|
28
|
+
const ORDER = ['case', 'arm', 'run'];
|
|
29
|
+
const SCOPES = {
|
|
30
|
+
case: (r, c) => c.counts,
|
|
31
|
+
arm: (r, c) => { const a = r.run && r.run.arms && r.run.arms[c.mode]; return a && a.counts; },
|
|
32
|
+
run: (r) => r.run && r.run.counts,
|
|
33
|
+
};
|
|
34
|
+
|
|
35
|
+
// The count of `kind` for results.cases[index], or null when the receipt does not
|
|
36
|
+
// establish it. run.judge.samples is the run-scope judge count (R-5), read only where
|
|
37
|
+
// run.counts is silent and only for an arm generated in this run.
|
|
38
|
+
function resolveCount(receipt, index, kind) {
|
|
39
|
+
if (!KINDS.includes(kind)) throw new Error(`unknown count kind: ${kind}`);
|
|
40
|
+
const c = receipt.results.cases[index];
|
|
41
|
+
for (const scope of ORDER) {
|
|
42
|
+
const at = SCOPES[scope](receipt, c);
|
|
43
|
+
if (has(at, kind)) return at[kind];
|
|
44
|
+
if (scope === 'run' && kind === 'judge_samples_per_generation' && receipt.run && receipt.run.judge && Number.isInteger(receipt.run.judge.samples)) return receipt.run.judge.samples;
|
|
45
|
+
if (scope === 'arm' && archived(receipt, c.mode)) return null;
|
|
46
|
+
}
|
|
47
|
+
return null;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
// Which model generated an arm, and when: the arm's override, else the run's.
|
|
51
|
+
function resolveArm(receipt, mode) {
|
|
52
|
+
const a = (receipt.run && receipt.run.arms && receipt.run.arms[mode]) || {};
|
|
53
|
+
return {
|
|
54
|
+
model_id: a.model_id || receipt.run.model_id,
|
|
55
|
+
generated_at: a.generated_at || receipt.run.generated_at || null,
|
|
56
|
+
archived: archived(receipt, mode),
|
|
57
|
+
};
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
// Write counts at the narrowest honest scope (spec 043 AC-3). `perCase` is one entry per
|
|
61
|
+
// results.cases index: { index, counts: { <kind>: integer } } with a kind left out when
|
|
62
|
+
// the writer cannot establish it. Mutates and returns the receipt; call before sealing.
|
|
63
|
+
function placeCounts(receipt, perCase) {
|
|
64
|
+
const cases = receipt.results.cases;
|
|
65
|
+
const fresh = perCase.filter((p) => !archived(receipt, cases[p.index].mode));
|
|
66
|
+
const put = (obj, kind, v) => { obj.counts = { ...(obj.counts || {}), [kind]: v }; };
|
|
67
|
+
for (const kind of KINDS) {
|
|
68
|
+
const entries = fresh.map((p) => ({ p, v: p.counts && Number.isInteger(p.counts[kind]) ? p.counts[kind] : null }));
|
|
69
|
+
if (!entries.length) continue;
|
|
70
|
+
const allKnown = entries.length === cases.filter((c) => !archived(receipt, c.mode)).length && entries.every((e) => e.v !== null);
|
|
71
|
+
const values = new Set(entries.map((e) => e.v));
|
|
72
|
+
if (allKnown && values.size === 1) { receipt.run.counts = { ...(receipt.run.counts || {}), [kind]: entries[0].v }; continue; }
|
|
73
|
+
const byMode = new Map();
|
|
74
|
+
for (const e of entries) { const m = cases[e.p.index].mode; if (!byMode.has(m)) byMode.set(m, []); byMode.get(m).push(e); }
|
|
75
|
+
for (const [mode, es] of byMode) {
|
|
76
|
+
const modeKnown = allKnown && es.length === cases.filter((c) => c.mode === mode).length;
|
|
77
|
+
if (modeKnown && new Set(es.map((e) => e.v)).size === 1) {
|
|
78
|
+
receipt.run.arms = receipt.run.arms || {};
|
|
79
|
+
receipt.run.arms[mode] = receipt.run.arms[mode] || {};
|
|
80
|
+
put(receipt.run.arms[mode], kind, es[0].v);
|
|
81
|
+
} else {
|
|
82
|
+
for (const e of es) if (e.v !== null) put(cases[e.p.index], kind, e.v);
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
return receipt;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
// The judge-sample count, or null when the receipt does not establish one at run scope.
|
|
90
|
+
function judgeSamplesOrNull(receipt) {
|
|
91
|
+
const j = receipt && receipt.run && receipt.run.judge;
|
|
92
|
+
if (j && Number.isInteger(j.samples)) return j.samples;
|
|
93
|
+
const c = receipt && receipt.run && receipt.run.counts;
|
|
94
|
+
return has(c, 'judge_samples_per_generation') ? c.judge_samples_per_generation : null;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
// v0.8 loader rule (spec 043 AC-2): the two spellings of the run's judge count agree.
|
|
98
|
+
function judgeCountsDisagree(receipt) {
|
|
99
|
+
const j = receipt && receipt.run && receipt.run.judge;
|
|
100
|
+
const c = receipt && receipt.run && receipt.run.counts;
|
|
101
|
+
return !!(j && Number.isInteger(j.samples) && has(c, 'judge_samples_per_generation') && c.judge_samples_per_generation !== j.samples);
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
module.exports = { KINDS, ORDER, resolveCount, resolveArm, placeCounts, judgeSamplesOrNull, judgeCountsDisagree };
|