driftproof 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +81 -16
- package/bin/driftproof +118 -2
- package/config.js +1 -1
- package/lib/decision.js +425 -0
- package/lib/runner.js +35 -3
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -158,6 +158,22 @@ npx driftproof <cmd> # no install — always the published version
|
|
|
158
158
|
npm install -g driftproof # or install the CLI globally
|
|
159
159
|
```
|
|
160
160
|
|
|
161
|
+
Inside Claude Code, there is a third path — the same CLI, reached from a slash
|
|
162
|
+
command:
|
|
163
|
+
|
|
164
|
+
```bash
|
|
165
|
+
claude plugin marketplace add driftproofhq/driftproof
|
|
166
|
+
claude plugin install driftproof@driftproofhq
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
That installs `/driftproof:init`, `/driftproof:run` and `/driftproof:badge`.
|
|
170
|
+
The CLI is the product and it runs on every surface Driftproof measures, while
|
|
171
|
+
the plugin is one install path for Claude Code users: it builds one argument
|
|
172
|
+
vector, hands it to the pinned runner, and writes the receipt that runner would
|
|
173
|
+
have written from the same arguments. The plugin's version is the runner version
|
|
174
|
+
it pins, and `/driftproof:run` spends your Claude Code subscription rather than
|
|
175
|
+
an API key. It measures; it never edits a skill.
|
|
176
|
+
|
|
161
177
|
The only runtime dependency is `ajv` (schema validation); `@anthropic-ai/sdk` is
|
|
162
178
|
optional and pulled in only for `CLAUDE_PROVIDER=api`. The CLI resolves its spec,
|
|
163
179
|
schema, and model registry from inside the package, so it runs the same from a
|
|
@@ -271,7 +287,7 @@ A receipt is the unit of evidence — one JSON document conforming to
|
|
|
271
287
|
"model_release_date": "2025-10-01",
|
|
272
288
|
"provider": "anthropic",
|
|
273
289
|
"surface": "claude-cli",
|
|
274
|
-
"runner_version": "0.
|
|
290
|
+
"runner_version": "0.10.0",
|
|
275
291
|
"date_utc": "2026-07-27T…Z",
|
|
276
292
|
"registry": "registered",
|
|
277
293
|
"transcripts": "hashes-only",
|
|
@@ -349,29 +365,78 @@ jobs:
|
|
|
349
365
|
runs-on: ubuntu-latest
|
|
350
366
|
steps:
|
|
351
367
|
- uses: actions/checkout@v4
|
|
352
|
-
- uses: driftproofhq/driftproof@v0.
|
|
368
|
+
- uses: driftproofhq/driftproof@v0.10.0
|
|
353
369
|
with:
|
|
354
370
|
skill-dir: skills/my-skill
|
|
355
371
|
models: claude-haiku-4-5
|
|
356
372
|
api-key: ${{ secrets.ANTHROPIC_API_KEY }}
|
|
357
373
|
# max-usd: <n> # override the dollar budget (default: DEV_MAX_USD in config.js)
|
|
358
374
|
# max-calls: <n> # override the per-model call cap (default: DEV_MAX_CALLS)
|
|
359
|
-
# fail-on-regression: 'true' # (default) fail the job
|
|
375
|
+
# fail-on-regression: 'true' # (default) fail the job on a measured regression
|
|
376
|
+
# (a requested model with no readable receipt fails the job either way)
|
|
360
377
|
```
|
|
361
378
|
|
|
362
|
-
The action runs the suite
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
is
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
379
|
+
The action runs the suite on **every** id in `models`, writes one receipt per
|
|
380
|
+
model, uploads the whole receipt directory as a build artifact, and renders the
|
|
381
|
+
run on three surfaces: the **badge**, a **job summary** carrying one row per
|
|
382
|
+
requested model, and the **check title** of the enforcement step.
|
|
383
|
+
|
|
384
|
+
**The decision is taken over every receipt the run produced.** Each requested
|
|
385
|
+
model gets one decision state, and the run's `verdict` is the **worst** of them —
|
|
386
|
+
worst first, in this order:
|
|
387
|
+
|
|
388
|
+
| `verdict` | decision state | when |
|
|
389
|
+
|---|---|---|
|
|
390
|
+
| `REGRESSED` | regression | measured, and the skill hurt: `delta <= -EFFECT_FLOOR` |
|
|
391
|
+
| `REFUSED` | refused | a requested model produced **no readable receipt** |
|
|
392
|
+
| `INCONCLUSIVE` | inconclusive | the run did not complete, or carries no numeric delta |
|
|
393
|
+
| `NOT_MEASURED` | not measured | the receipt is below `TESTED`, or nothing in it says a model answered |
|
|
394
|
+
| `NO_EFFECT` | no detected effect | measured, and `\|delta\| < EFFECT_FLOOR` |
|
|
395
|
+
| `PASSED` | helped | measured, and the skill helped: `delta >= EFFECT_FLOOR` |
|
|
396
|
+
|
|
397
|
+
So on a multi-model run a regression on **any** one of them governs the verdict,
|
|
398
|
+
and the badge names the model it came from. Earlier releases decided the job from
|
|
399
|
+
whichever receipt's filename sorted **first**, which is a property of the model id
|
|
400
|
+
and not of the run — so if a Driftproof check was ever green on a multi-model run,
|
|
401
|
+
that result was only ever about one of your models, and is worth re-running before
|
|
402
|
+
you rely on it.
|
|
403
|
+
|
|
404
|
+
**What fails the job.** `fail-on-regression` governs **regression verdicts
|
|
405
|
+
only**: set it to `'false'` and a measured regression warns instead of failing.
|
|
406
|
+
`REFUSED` is not a verdict about your skill — it is the run failing to produce
|
|
407
|
+
one for a model you asked for — so it **fails the job whatever
|
|
408
|
+
`fail-on-regression` says**. A workflow that sets `fail-on-regression: 'false'`
|
|
409
|
+
to keep the check non-blocking can still be failed this way, deliberately: a
|
|
410
|
+
model that was not measured did not pass. `INCONCLUSIVE` and `NOT_MEASURED` do
|
|
411
|
+
not fail the job on their own — a run that could not measure is not evidence that
|
|
412
|
+
the skill hurt — but they **never render as success** on any of the three
|
|
413
|
+
surfaces: not in the badge, not in the summary row, and not in the check title,
|
|
414
|
+
which carries a `::warning` naming the state and the models it came from.
|
|
415
|
+
|
|
416
|
+
Step outputs: `verdict`, `delta` (of the model the worst decision came from),
|
|
417
|
+
`worst_state`, `regressed_models`, `missing_models` and `receipts_dir`. Each is
|
|
418
|
+
described in [`action.yml`](action.yml).
|
|
419
|
+
|
|
420
|
+
For a free CI dry-run with **zero model calls**, set `DRIFTPROOF_STUB=1` in the job
|
|
421
|
+
env — the runner returns canned receipts so the wiring can be tested without spend
|
|
422
|
+
(this is exactly how the action's own
|
|
423
|
+
[self-test](.github/workflows/action-selftest.yml) runs). A stub receipt says what
|
|
424
|
+
it is: `verification_level` `UNVERIFIED`, `run.surface` `stub`,
|
|
425
|
+
`run.answered_by.kind` `stub`, and the verdict on it is `NOT_MEASURED` — it proves
|
|
426
|
+
the wiring, never the skill, and the badge, the summary row and the check title
|
|
427
|
+
all say so rather than exiting quietly green.
|
|
428
|
+
|
|
429
|
+
That self-test is also the proof behind the **Action's** input hardening, and it
|
|
430
|
+
is worth saying exactly which surface that covers. On every CI run the Action
|
|
431
|
+
self-test sends one hostile value through each of the Action's five inputs — a
|
|
432
|
+
quote, a semicolon, `$(...)`, a backtick and a newline — on the real runner, and
|
|
433
|
+
the Action fails unless every one of them is refused before anything ran, so a
|
|
434
|
+
green check there is a claim you can read rather than one you have to take.
|
|
435
|
+
Those refusals live in `action/lib.sh`, and what they protect is the Action
|
|
436
|
+
surface. The CLI has carried its own input contract at its own door since
|
|
437
|
+
0.9.0 — the same rules, held identical to the Action's by the gate — so
|
|
438
|
+
`npx driftproof` refuses a malformed cap or model id before it projects a run.
|
|
439
|
+
Neither statement covers the other. Said of the Action, of CI, or of the runner, "hostile input is refused" is only ever true of the one surface it was measured on.
|
|
375
440
|
|
|
376
441
|
### Badge
|
|
377
442
|
|
package/bin/driftproof
CHANGED
|
@@ -14,6 +14,7 @@ const { surfaceForModel, isSubscriptionSurface, resolveModel, evalUser } = requi
|
|
|
14
14
|
const { estimateRunCostUSD, BudgetTracker } = require('../lib/cost');
|
|
15
15
|
const { registryStatus, assertRegistered, REGISTRY_PATH } = require('../lib/models');
|
|
16
16
|
const { verdictFromReceipt, badgeEndpoint, githubOutputLines } = require('../lib/verdict');
|
|
17
|
+
const decision = require('../lib/decision');
|
|
17
18
|
const { scaffoldInit } = require('../lib/init');
|
|
18
19
|
|
|
19
20
|
// Output dirs default to the USER's current directory, not the package dir, so a
|
|
@@ -176,7 +177,9 @@ USAGE
|
|
|
176
177
|
[--keep-transcripts] [--out DIR] [--trusted-skill]
|
|
177
178
|
${PROJECT_NAME} diff <receiptA.json> <receiptB.json> [--out FILE] [--mode release|revision]
|
|
178
179
|
${PROJECT_NAME} validate <receipt.json>
|
|
179
|
-
${PROJECT_NAME} badge <receipt.json> [--out FILE] [--github-output]
|
|
180
|
+
${PROJECT_NAME} badge <receipt.json|receipts-dir> [--models a,b] [--out FILE] [--github-output]
|
|
181
|
+
${PROJECT_NAME} decide <receipts-dir> --models a,b [--github-output] [--badge FILE]
|
|
182
|
+
[--summary FILE] [--enforce] [--fail-on-regression true|false]
|
|
180
183
|
${PROJECT_NAME} import <results.json> --from agent-skills-eval|skillgrade [--out DIR]
|
|
181
184
|
${PROJECT_NAME} export <receipt.json> [--to summary-json] [--report-url URL] [--out FILE]
|
|
182
185
|
|
|
@@ -504,7 +507,12 @@ See AUTHORING.md for how to write a fair suite.`);
|
|
|
504
507
|
// --github-output prints verdict/delta/message/color as key=value lines.
|
|
505
508
|
function cmdBadge(positional, flags) {
|
|
506
509
|
const p = positional[0];
|
|
507
|
-
if (!p) { console.error('usage: driftproof badge <receipt.json> [--out FILE] [--github-output]'); process.exit(2); }
|
|
510
|
+
if (!p) { console.error('usage: driftproof badge <receipt.json|receipts-dir> [--models a,b] [--out FILE] [--github-output]'); process.exit(2); }
|
|
511
|
+
// SPEC 030 AC-3. A DIRECTORY renders the worst decision across every receipt
|
|
512
|
+
// in it, naming the model that state came from. A single receipt renders
|
|
513
|
+
// exactly what it always did - lib/verdict.js is untouched and still answers
|
|
514
|
+
// the one-receipt question, which AC-3's control asserts.
|
|
515
|
+
if (fs.existsSync(p) && fs.statSync(p).isDirectory()) return cmdBadgeSet(p, flags);
|
|
508
516
|
const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
509
517
|
refuseUnverified(receipt, p, 'badge');
|
|
510
518
|
if (flags['github-output']) { console.log(githubOutputLines(receipt)); return; }
|
|
@@ -521,6 +529,113 @@ function cmdBadge(positional, flags) {
|
|
|
521
529
|
}
|
|
522
530
|
}
|
|
523
531
|
|
|
532
|
+
// The badge over a receipt SET (spec 030 AC-3): the worst decision, the model
|
|
533
|
+
// it came from, and how many models share it. Every receipt in the directory is
|
|
534
|
+
// read; when --models is given the requested list governs, so a model with no
|
|
535
|
+
// receipt is `refused` and can be the worst state.
|
|
536
|
+
function cmdBadgeSet(dir, flags) {
|
|
537
|
+
const files = decision.receiptFiles(dir);
|
|
538
|
+
if (files.length === 0) {
|
|
539
|
+
console.error(` \u2717 REFUSED (badge): ${dir} holds no receipts; a badge over nothing is a badge about nothing.`);
|
|
540
|
+
process.exit(2);
|
|
541
|
+
}
|
|
542
|
+
// Every receipt is verified before it is displayed, exactly as the
|
|
543
|
+
// single-receipt path does (spec 026 AC-20): a set is not a way around the
|
|
544
|
+
// check, and the file that fails is named.
|
|
545
|
+
for (const f of files) {
|
|
546
|
+
const r = JSON.parse(fs.readFileSync(f, 'utf8'));
|
|
547
|
+
refuseUnverified(r, f, 'badge');
|
|
548
|
+
}
|
|
549
|
+
const models = flags.models || files
|
|
550
|
+
.map((f) => JSON.parse(fs.readFileSync(f, 'utf8')))
|
|
551
|
+
.map((r) => r.run && r.run.model_id).filter(Boolean).join(',');
|
|
552
|
+
const d = decision.decideSet(dir, models);
|
|
553
|
+
if (flags['github-output']) { console.log(decision.githubOutputLines(d)); return; }
|
|
554
|
+
const badge = decision.badgeEndpointForSet(d);
|
|
555
|
+
const json = JSON.stringify(badge, null, 2);
|
|
556
|
+
if (flags.out) {
|
|
557
|
+
const out = path.resolve(flags.out);
|
|
558
|
+
fs.mkdirSync(path.dirname(out), { recursive: true });
|
|
559
|
+
fs.writeFileSync(out, json + '\n');
|
|
560
|
+
console.log(`badge written to ${flags.out} (${d.worst}: ${badge.message}, ${badge.color})`);
|
|
561
|
+
} else {
|
|
562
|
+
console.log(json);
|
|
563
|
+
}
|
|
564
|
+
}
|
|
565
|
+
|
|
566
|
+
// The decision over a RUN: every receipt it produced, against every model it was
|
|
567
|
+
// asked for (spec 030 AC-1, AC-2, AC-4).
|
|
568
|
+
//
|
|
569
|
+
// `badge` answers about one receipt and still does. This answers about the set,
|
|
570
|
+
// which is what the GitHub Action needs and what `ls … | head -1` was standing
|
|
571
|
+
// in for. Modes compose: --github-output emits the step outputs, --summary
|
|
572
|
+
// writes the job-summary table, --enforce decides the job and sets the exit
|
|
573
|
+
// code. With none of them it prints the decision as JSON.
|
|
574
|
+
function cmdDecide(positional, flags) {
|
|
575
|
+
const dir = positional[0];
|
|
576
|
+
if (!dir || !flags.models) {
|
|
577
|
+
console.error('usage: driftproof decide <receipts-dir> --models a,b [--github-output] [--badge FILE] [--summary FILE] [--enforce] [--fail-on-regression true|false]');
|
|
578
|
+
process.exit(2);
|
|
579
|
+
}
|
|
580
|
+
if (!fs.existsSync(dir) || !fs.statSync(dir).isDirectory()) {
|
|
581
|
+
console.error(` \u2717 REFUSED (decide): ${dir} is not a directory; a run that wrote no receipt directory decided nothing.`);
|
|
582
|
+
process.exit(2);
|
|
583
|
+
}
|
|
584
|
+
// ── THE SEAL IS VERIFIED BEFORE ANY DECISION (spec 026 AC-20, spec 030 AC-3)
|
|
585
|
+
//
|
|
586
|
+
// Before spec 030 the action rendered through `driftproof badge <receipt>`,
|
|
587
|
+
// which calls refuseUnverified and exits 4 on a receipt_hash mismatch. All
|
|
588
|
+
// four surfaces the action writes now go through THIS command, and until
|
|
589
|
+
// this it verified nothing: a receipt with one field hand-edited after
|
|
590
|
+
// sealing rendered `passing`, `brightgreen`, exit 0 - the check that had
|
|
591
|
+
// stood between the receipts and the job was not carried across to the
|
|
592
|
+
// command that replaced its caller. The failure direction is FAIL-OPEN,
|
|
593
|
+
// which is the one direction an adopter cannot detect from their own CI.
|
|
594
|
+
//
|
|
595
|
+
// Every receipt in the set, before anything is read for a decision and
|
|
596
|
+
// before any file is written, exactly as cmdBadgeSet does - a set is not a
|
|
597
|
+
// way around the check, and the file that fails is named.
|
|
598
|
+
//
|
|
599
|
+
// A receipt that cannot be PARSED is left to the decision, not refused here:
|
|
600
|
+
// `decideSet` distinguishes a receipt that is absent from one that exists and
|
|
601
|
+
// is unreadable and fails closed on both, naming which (spec 030 AC-2,
|
|
602
|
+
// absence-vs-unreadable). Refusing it here would collapse that distinction
|
|
603
|
+
// into an exit code.
|
|
604
|
+
for (const f of decision.receiptFiles(dir)) {
|
|
605
|
+
let receipt;
|
|
606
|
+
try { receipt = JSON.parse(fs.readFileSync(f, 'utf8')); } catch (_e) { continue; }
|
|
607
|
+
refuseUnverified(receipt, f, 'decide');
|
|
608
|
+
}
|
|
609
|
+
const failOnRegression = String(flags['fail-on-regression'] ?? 'true') !== 'false';
|
|
610
|
+
const d = decision.decideSet(dir, flags.models, { failOnRegression });
|
|
611
|
+
if (d.rows.length === 0) {
|
|
612
|
+
console.error(' \u2717 REFUSED (decide): no models requested; there is nothing to decide over.');
|
|
613
|
+
process.exit(2);
|
|
614
|
+
}
|
|
615
|
+
let acted = false;
|
|
616
|
+
if (flags['github-output']) { console.log(decision.githubOutputLines(d)); acted = true; }
|
|
617
|
+
if (flags.badge) {
|
|
618
|
+
const out = path.resolve(flags.badge);
|
|
619
|
+
fs.mkdirSync(path.dirname(out), { recursive: true });
|
|
620
|
+
const badge = decision.badgeEndpointForSet(d);
|
|
621
|
+
fs.writeFileSync(out, JSON.stringify(badge, null, 2) + '\n');
|
|
622
|
+
console.log(`badge written to ${flags.badge} (${d.worst}: ${badge.message}, ${badge.color})`);
|
|
623
|
+
acted = true;
|
|
624
|
+
}
|
|
625
|
+
if (flags.summary) {
|
|
626
|
+
const out = path.resolve(flags.summary);
|
|
627
|
+
fs.mkdirSync(path.dirname(out), { recursive: true });
|
|
628
|
+
fs.appendFileSync(out, decision.summaryMarkdown(d) + '\n');
|
|
629
|
+
acted = true;
|
|
630
|
+
}
|
|
631
|
+
if (flags.enforce) {
|
|
632
|
+
for (const line of decision.enforcementLines(d, { failOnRegression })) console.log(line);
|
|
633
|
+
acted = true;
|
|
634
|
+
if (d.fails) process.exit(1);
|
|
635
|
+
}
|
|
636
|
+
if (!acted) console.log(JSON.stringify(d, null, 2));
|
|
637
|
+
}
|
|
638
|
+
|
|
524
639
|
// Convert another tool's results file into a valid DECLARED receipt (interop,
|
|
525
640
|
// Phase 7). The converted receipt is validated + self-hash-verified before it
|
|
526
641
|
// is written; a failed conversion writes nothing.
|
|
@@ -579,6 +694,7 @@ async function main() {
|
|
|
579
694
|
case 'diff': return cmdDiff(positional, flags);
|
|
580
695
|
case 'validate': return cmdValidate(positional);
|
|
581
696
|
case 'badge': return cmdBadge(positional, flags);
|
|
697
|
+
case 'decide': return cmdDecide(positional, flags);
|
|
582
698
|
case 'import': return cmdImport(positional, flags);
|
|
583
699
|
case 'export': return cmdExport(positional, flags);
|
|
584
700
|
case 'version': case '--version': case '-v': console.log(RUNNER_VERSION); return;
|
package/config.js
CHANGED
|
@@ -9,7 +9,7 @@ const PROJECT_NAME = 'driftproof';
|
|
|
9
9
|
// Bumped whenever the runner's behaviour or receipt-generation semantics change
|
|
10
10
|
// in a way that could affect results. Recorded into every receipt as
|
|
11
11
|
// run.runner_version so a receipt is reproducible against a known engine.
|
|
12
|
-
const RUNNER_VERSION = '0.
|
|
12
|
+
const RUNNER_VERSION = '0.10.0';
|
|
13
13
|
|
|
14
14
|
// The eval format we CONSUME (we deliberately do not invent our own).
|
|
15
15
|
const SUITE_FORMAT = 'agentskills.io/evals';
|
package/lib/decision.js
ADDED
|
@@ -0,0 +1,425 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// The decision a RUN reaches, over every receipt it produced (spec 030).
|
|
5
|
+
//
|
|
6
|
+
// `lib/verdict.js` answers a question about ONE receipt: does the skill still
|
|
7
|
+
// help on THIS model? That question is still asked and still answered there,
|
|
8
|
+
// unchanged. This module answers the one the action actually needs and never
|
|
9
|
+
// had a place to ask: a run takes a LIST of models and must reach a single
|
|
10
|
+
// decision over all of them.
|
|
11
|
+
//
|
|
12
|
+
// Until spec 030, `action/run.sh` picked one receipt with `ls … | head -1` and
|
|
13
|
+
// decided the job from it. Receipts are named `<skill>-<model>-<date>.json`, so
|
|
14
|
+
// that pick is ordered by model id: `claude-haiku-4-5` sorts before
|
|
15
|
+
// `claude-sonnet-5`, and a regression on any model that did not sort first was
|
|
16
|
+
// reported as a pass, with a green check and a brightgreen badge. The seam lives
|
|
17
|
+
// here, in lib/, rather than in the shell, because `bin/driftproof badge` needs
|
|
18
|
+
// it too and because a decision state is a property of a receipt.
|
|
19
|
+
|
|
20
|
+
const fs = require('fs');
|
|
21
|
+
const path = require('path');
|
|
22
|
+
const { EFFECT_FLOOR } = require('../config');
|
|
23
|
+
const { shortModel, githubOutputEntry } = require('./verdict');
|
|
24
|
+
|
|
25
|
+
// The six decision states (spec 030 AC-4).
|
|
26
|
+
//
|
|
27
|
+
// `helped` is the sixth, and is a RECORDED DEVIATION from the brief's five
|
|
28
|
+
// (spec.md A-030-1): a model on which the skill measurably helped needs a row
|
|
29
|
+
// that says so. `no detected effect` is false about such a model - the effect
|
|
30
|
+
// was detected and it cleared the floor - and widening it to mean "not a
|
|
31
|
+
// regression" would put a measured lift and a measured nothing in one cell.
|
|
32
|
+
const STATES = ['regression', 'refused', 'inconclusive', 'not measured', 'no detected effect', 'helped'];
|
|
33
|
+
|
|
34
|
+
// WORST FIRST. This is the single definition of the ordering; AC-1's
|
|
35
|
+
// enforcement and AC-3's badge both read it, so they cannot disagree about
|
|
36
|
+
// which of two states governs a run.
|
|
37
|
+
const STATE_ORDER = STATES.slice();
|
|
38
|
+
|
|
39
|
+
// States that must never render as success on any surface the action writes -
|
|
40
|
+
// the badge, the summary row, or the check title (AC-4).
|
|
41
|
+
const NEVER_SUCCESS = ['inconclusive', 'not measured'];
|
|
42
|
+
|
|
43
|
+
// States that fail the job. `refused` fails CLOSED (AC-2): a model that
|
|
44
|
+
// produced no receipt is not a model that passed. `regression` fails subject to
|
|
45
|
+
// fail-on-regression. An unmeasured run does not fail on its own (AC-4's rule)
|
|
46
|
+
// - it is not evidence that the skill hurt - but it never renders as success.
|
|
47
|
+
function failsJob(state, { failOnRegression = true } = {}) {
|
|
48
|
+
// AC-2. `refused` fails CLOSED and is NOT subject to fail-on-regression: that
|
|
49
|
+
// input says what to do about a measured regression, and a model that
|
|
50
|
+
// produced no receipt was not measured at all. A run that silently narrowed
|
|
51
|
+
// to the models that survived would be AC-1's defect with a different cause -
|
|
52
|
+
// the decision taken over a subset nobody chose.
|
|
53
|
+
if (state === 'refused') return true;
|
|
54
|
+
if (state === 'regression') return !!failOnRegression;
|
|
55
|
+
return false;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
// The mapping rule of spec 030 AC-4, in order; the first match wins. A null or
|
|
59
|
+
// missing receipt is `refused`.
|
|
60
|
+
//
|
|
61
|
+
// `not measured` is tested BEFORE `inconclusive` and before any delta is read,
|
|
62
|
+
// because a receipt below TESTED has not measured anything its delta could be
|
|
63
|
+
// about - spec 026 AC-1 is what makes a stub receipt unable to claim TESTED,
|
|
64
|
+
// and reading its delta first would grade a run that never ran.
|
|
65
|
+
// The rule is read off AC-4's table and nothing is supplied that the receipt
|
|
66
|
+
// does not carry. Until this it read
|
|
67
|
+
//
|
|
68
|
+
// const level = receipt.verification_level || 'TESTED';
|
|
69
|
+
// ... || (kind && kind !== 'model')
|
|
70
|
+
//
|
|
71
|
+
// - a receipt with no `verification_level` was GRADED ON ITS DELTA as though it
|
|
72
|
+
// had claimed TESTED, and a receipt with no `answered_by` block had the second
|
|
73
|
+
// clause skipped entirely. Both defaults point the same way: they let a receipt
|
|
74
|
+
// that says nothing about how it was answered reach `helped`. AC-4 says each
|
|
75
|
+
// state SHALL be derived "by the rule below and by no other", and the rule says
|
|
76
|
+
// `verification_level != "TESTED"` and `run.answered_by.kind != "model"` - an
|
|
77
|
+
// absent field satisfies both. The receipts the shipped runner writes always
|
|
78
|
+
// carry a level and an answered_by block (lib/run.js answeredByOf always
|
|
79
|
+
// returns a kind), so this changes nothing about a run the action produced; it
|
|
80
|
+
// changes what happens to a receipt from somewhere else, and it changes it in
|
|
81
|
+
// the fail-safe direction.
|
|
82
|
+
function decisionState(receipt) {
|
|
83
|
+
if (!receipt) return 'refused';
|
|
84
|
+
const level = receipt.verification_level;
|
|
85
|
+
const kind = receipt.run && receipt.run.answered_by ? receipt.run.answered_by.kind : undefined;
|
|
86
|
+
if (level !== 'TESTED' || kind !== 'model') return 'not measured';
|
|
87
|
+
const cmp = receipt.comparison || {};
|
|
88
|
+
if ((receipt.run && receipt.run.status === 'incomplete') || typeof cmp.delta !== 'number') return 'inconclusive';
|
|
89
|
+
if (cmp.delta <= -EFFECT_FLOOR) return 'regression';
|
|
90
|
+
if (cmp.delta >= EFFECT_FLOOR) return 'helped';
|
|
91
|
+
return 'no detected effect';
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
// The worst state in a set, by STATE_ORDER. An empty set has no decision, and
|
|
95
|
+
// says so with null rather than defaulting to something benign.
|
|
96
|
+
function worstState(states) {
|
|
97
|
+
let worst = null;
|
|
98
|
+
for (const s of states) {
|
|
99
|
+
const i = STATE_ORDER.indexOf(s);
|
|
100
|
+
if (i < 0) continue;
|
|
101
|
+
if (worst === null || i < STATE_ORDER.indexOf(worst)) worst = s;
|
|
102
|
+
}
|
|
103
|
+
return worst;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
// Every receipt in a directory, excluding the summaries and the badge writer's
|
|
107
|
+
// own earlier output. Mirrors what action/run.sh's glob selected from, so the
|
|
108
|
+
// set this reads is the set that was there to be read.
|
|
109
|
+
function receiptFiles(dir) {
|
|
110
|
+
return fs.readdirSync(dir)
|
|
111
|
+
.filter((f) => f.endsWith('.json') && !f.includes('.summary.') && f !== 'badge.json')
|
|
112
|
+
.sort()
|
|
113
|
+
.map((f) => path.join(dir, f));
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
// Parse the `models` input the same way the runner does: comma-separated, with
|
|
117
|
+
// surrounding whitespace ignored and empty entries dropped.
|
|
118
|
+
function parseModels(csv) {
|
|
119
|
+
return String(csv || '').split(',').map((s) => s.trim()).filter(Boolean);
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
// A receipt matches a requested id exactly, or after the release date stamp is
|
|
123
|
+
// stripped from both - `claude-haiku-4-5-20251001` IS `claude-haiku-4-5`, which
|
|
124
|
+
// is the equivalence lib/verdict.js's shortModel already defines for the badge.
|
|
125
|
+
function matchesModel(receiptModelId, requested) {
|
|
126
|
+
return receiptModelId === requested || shortModel(receiptModelId) === shortModel(requested);
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
// The same question for a receipt that CANNOT BE READ. `matchesModel` reads
|
|
130
|
+
// `run.model_id`, and for an unreadable receipt that field is inside the bytes
|
|
131
|
+
// that will not parse. The FILENAME is the only thing left that names a model:
|
|
132
|
+
// the runner writes `<skill>-<model>-<date>.json`, so the id appears in the
|
|
133
|
+
// basename as a dash-delimited run.
|
|
134
|
+
//
|
|
135
|
+
// Until this, the unreadable branch matched by POSITION - the first unreadable
|
|
136
|
+
// file in the directory, whichever model it belonged to. With one requested model
|
|
137
|
+
// absent and another's receipt unreadable, the two causes were hung on the wrong
|
|
138
|
+
// models: the absent model was reported as "exists and is unreadable" naming the
|
|
139
|
+
// OTHER model's file, and the model whose file was actually corrupt was reported
|
|
140
|
+
// as having produced nothing. Both still failed closed and both ids still
|
|
141
|
+
// appeared somewhere, which is why AC-2's arms stayed green - they drove one
|
|
142
|
+
// cause at a time and never both at once (F-1 of 2026-09-12).
|
|
143
|
+
function fileNamesModel(file, requested) {
|
|
144
|
+
const base = path.basename(file);
|
|
145
|
+
// Both spellings, for the reason `matchesModel` takes both: a receipt for
|
|
146
|
+
// `claude-haiku-4-5-20251001` answers a request for `claude-haiku-4-5`.
|
|
147
|
+
for (const id of new Set([requested, shortModel(requested)])) {
|
|
148
|
+
if (base.includes(`-${id}-`)) return true;
|
|
149
|
+
}
|
|
150
|
+
return false;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
// The decision over a run: one row per REQUESTED model, in the order requested.
|
|
154
|
+
//
|
|
155
|
+
// Rows come from the requested list rather than from the directory listing, so
|
|
156
|
+
// a model that produced no receipt is a row that says `refused` instead of a
|
|
157
|
+
// row that is simply absent. That is the whole of AC-2: absence is not a pass.
|
|
158
|
+
function decideSet(dir, requestedCsv, { failOnRegression = true } = {}) {
|
|
159
|
+
const requested = parseModels(requestedCsv);
|
|
160
|
+
const files = receiptFiles(dir);
|
|
161
|
+
const loaded = files.map((f) => {
|
|
162
|
+
let receipt = null, error = null;
|
|
163
|
+
try { receipt = JSON.parse(fs.readFileSync(f, 'utf8')); } catch (e) { error = e.message; }
|
|
164
|
+
return { file: f, receipt, error };
|
|
165
|
+
});
|
|
166
|
+
const used = new Set();
|
|
167
|
+
const rows = requested.map((model) => {
|
|
168
|
+
const hit = loaded.find((l) => !used.has(l.file) && l.receipt
|
|
169
|
+
&& l.receipt.run && matchesModel(l.receipt.run.model_id, model));
|
|
170
|
+
if (hit) used.add(hit.file);
|
|
171
|
+
// A receipt that exists and cannot be parsed is UNREADABLE, not absent.
|
|
172
|
+
// Both fail closed, and the row says which - the register's
|
|
173
|
+
// absence-vs-unreadable distinction, kept at the point it is decided.
|
|
174
|
+
const unreadable = !hit && loaded.find((l) => !used.has(l.file) && l.error
|
|
175
|
+
&& fileNamesModel(l.file, model));
|
|
176
|
+
if (unreadable) used.add(unreadable.file);
|
|
177
|
+
const receipt = hit ? hit.receipt : null;
|
|
178
|
+
const state = decisionState(receipt);
|
|
179
|
+
return {
|
|
180
|
+
model,
|
|
181
|
+
state,
|
|
182
|
+
delta: receipt && receipt.comparison && typeof receipt.comparison.delta === 'number'
|
|
183
|
+
? receipt.comparison.delta : null,
|
|
184
|
+
file: hit ? path.basename(hit.file) : (unreadable ? path.basename(unreadable.file) : null),
|
|
185
|
+
// The distinction is carried on the ROW, not left in a sentence, because
|
|
186
|
+
// the enforcement message has to say which of the two happened (AC-2's
|
|
187
|
+
// mutation class, absence-vs-unreadable).
|
|
188
|
+
unreadable: unreadable ? unreadable.error : null,
|
|
189
|
+
// A SHORT CAUSE, not the parser's text. V8's JSON.parse message echoes the
|
|
190
|
+
// first bytes of the input - "Unexpected token '|', \"|{bad\" is not valid
|
|
191
|
+
// JSON" - and this string is rendered into a markdown table cell, where one
|
|
192
|
+
// unescaped `|` adds a column and a newline ends the row. The detail still
|
|
193
|
+
// reaches the annotation, which is not a table; the cell carries the cause,
|
|
194
|
+
// and the filename is already beside it in the same cell (F-2 of
|
|
195
|
+
// 2026-09-12).
|
|
196
|
+
reason: unreadable ? 'receipt unreadable (not valid JSON)' : (hit ? null : 'no receipt for this model'),
|
|
197
|
+
};
|
|
198
|
+
});
|
|
199
|
+
// Receipts the run wrote for models nobody requested are reported rather than
|
|
200
|
+
// dropped: they are evidence that the run did something other than what it
|
|
201
|
+
// was asked for.
|
|
202
|
+
const unexpected = loaded.filter((l) => !used.has(l.file))
|
|
203
|
+
// An unreadable file whose name matches no requested model is still reported,
|
|
204
|
+
// and reported AS unreadable. Calling it a receipt for a model nobody asked
|
|
205
|
+
// for would state as fact the one thing its bytes could not tell us.
|
|
206
|
+
.map((l) => (l.error ? `${path.basename(l.file)} (unreadable)` : path.basename(l.file)));
|
|
207
|
+
const worst = worstState(rows.map((r) => r.state));
|
|
208
|
+
return {
|
|
209
|
+
rows,
|
|
210
|
+
worst,
|
|
211
|
+
regressed: rows.filter((r) => r.state === 'regression').map((r) => r.model),
|
|
212
|
+
missing: rows.filter((r) => r.state === 'refused').map((r) => r.model),
|
|
213
|
+
// `missing` is every model that reached no readable receipt, which is what
|
|
214
|
+
// the missing_models output has always published. `absent` and `unreadable`
|
|
215
|
+
// split it by CAUSE: nothing was written for this model, or something was
|
|
216
|
+
// written and cannot be read. Both fail closed; they are not the same fact,
|
|
217
|
+
// and the failure has to say which.
|
|
218
|
+
absent: rows.filter((r) => r.state === 'refused' && !r.unreadable).map((r) => r.model),
|
|
219
|
+
unreadable: rows.filter((r) => r.state === 'refused' && r.unreadable)
|
|
220
|
+
.map((r) => ({ model: r.model, file: r.file, error: r.unreadable })),
|
|
221
|
+
unexpected,
|
|
222
|
+
receiptCount: files.length,
|
|
223
|
+
requestedCount: requested.length,
|
|
224
|
+
fails: rows.some((r) => failsJob(r.state, { failOnRegression })),
|
|
225
|
+
};
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
module.exports = {
|
|
229
|
+
STATES, STATE_ORDER, NEVER_SUCCESS, EFFECT_FLOOR,
|
|
230
|
+
decisionState, worstState, failsJob, decideSet, receiptFiles, parseModels, matchesModel,
|
|
231
|
+
};
|
|
232
|
+
|
|
233
|
+
// ── the three surfaces the action writes ────────────────────────────────────
|
|
234
|
+
//
|
|
235
|
+
// AC-4 binds all three: the badge, the summary row and the check title. A state
|
|
236
|
+
// in NEVER_SUCCESS must not render as success on any of them, which is why the
|
|
237
|
+
// renderers live together - three surfaces fed by one table cannot drift apart
|
|
238
|
+
// the way three hand-written strings do.
|
|
239
|
+
|
|
240
|
+
// How each state renders. `color` is the shields colour; `marker` is the
|
|
241
|
+
// summary row's marker. Only `helped` gets a success marker: `no detected
|
|
242
|
+
// effect` is not a failure but it is not a success either, and the two
|
|
243
|
+
// NEVER_SUCCESS states carry a warning.
|
|
244
|
+
const RENDER = {
|
|
245
|
+
regression: { word: 'regressed', color: 'red', marker: '\u274c' },
|
|
246
|
+
refused: { word: 'refused', color: 'red', marker: '\u274c' },
|
|
247
|
+
inconclusive: { word: 'inconclusive', color: 'yellow', marker: '\u26a0\ufe0f' },
|
|
248
|
+
'not measured': { word: 'not measured', color: 'lightgrey', marker: '\u26a0\ufe0f' },
|
|
249
|
+
'no detected effect': { word: 'no effect', color: 'lightgrey', marker: '\u2014' },
|
|
250
|
+
helped: { word: 'passing', color: 'brightgreen', marker: '\u2705' },
|
|
251
|
+
};
|
|
252
|
+
|
|
253
|
+
// The one place a state becomes a success claim. AC-4's clause is enforced here
|
|
254
|
+
// rather than restated at each surface.
|
|
255
|
+
function rendersAsSuccess(state) {
|
|
256
|
+
return state === 'helped';
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
// The badge over a SET (AC-3): the worst state, naming the model it came from,
|
|
260
|
+
// and how many models share it when more than one does.
|
|
261
|
+
function badgeMessage(d) {
|
|
262
|
+
const worst = d.worst;
|
|
263
|
+
if (!worst) return 'no receipts';
|
|
264
|
+
const r = RENDER[worst];
|
|
265
|
+
const sharing = d.rows.filter((row) => row.state === worst);
|
|
266
|
+
const named = sharing[0] ? sharing[0].model : 'unknown';
|
|
267
|
+
const extra = sharing.length > 1 ? ` (+${sharing.length - 1} more)` : '';
|
|
268
|
+
return `${r.word} on ${shortModel(named)}${extra}`;
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
function badgeEndpointForSet(d, { label = 'driftproof' } = {}) {
|
|
272
|
+
const worst = d.worst;
|
|
273
|
+
return {
|
|
274
|
+
schemaVersion: 1,
|
|
275
|
+
label,
|
|
276
|
+
message: badgeMessage(d),
|
|
277
|
+
color: worst ? RENDER[worst].color : 'lightgrey',
|
|
278
|
+
};
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
// The step outputs. `verdict` keeps the vocabulary lib/verdict.js publishes, so
|
|
282
|
+
// a workflow reading steps.run.outputs.verdict is not broken by this change;
|
|
283
|
+
// `worst`, `regressed` and `missing` are new and carry what it could not say.
|
|
284
|
+
const VERDICT_WORD = {
|
|
285
|
+
regression: 'REGRESSED', refused: 'REFUSED', inconclusive: 'INCONCLUSIVE',
|
|
286
|
+
'not measured': 'NOT_MEASURED', 'no detected effect': 'NO_EFFECT', helped: 'PASSED',
|
|
287
|
+
};
|
|
288
|
+
|
|
289
|
+
function githubOutputLines(d) {
|
|
290
|
+
const worst = d.worst;
|
|
291
|
+
const badge = badgeEndpointForSet(d);
|
|
292
|
+
const worstRow = d.rows.find((r) => r.state === worst);
|
|
293
|
+
return [
|
|
294
|
+
['verdict', worst ? VERDICT_WORD[worst] : 'NOT_MEASURED'],
|
|
295
|
+
['delta', worstRow && worstRow.delta !== null ? worstRow.delta : 0],
|
|
296
|
+
['message', badge.message],
|
|
297
|
+
['color', badge.color],
|
|
298
|
+
['worst_state', worst || 'none'],
|
|
299
|
+
['regressed_models', d.regressed.join(',')],
|
|
300
|
+
['missing_models', d.missing.join(',')],
|
|
301
|
+
['receipt_count', d.receiptCount],
|
|
302
|
+
].map(([k, v]) => githubOutputEntry(k, v)).join('\n');
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
// Anything reaching a table cell from outside this module - a filename, a reason,
|
|
306
|
+
// a model id - is escaped FOR THE CELL it sits in. A markdown row is delimited by
|
|
307
|
+
// `|`, so one unescaped pipe adds a column and the row stops lining up with its
|
|
308
|
+
// header; a newline ends the row outright. The cause AC-4's own finding had was a
|
|
309
|
+
// parser echo, and that echo is gone from the cell - but the filename is still
|
|
310
|
+
// read off a directory, so the row is made to survive the byte rather than made
|
|
311
|
+
// to depend on where the byte came from.
|
|
312
|
+
function cell(s) {
|
|
313
|
+
return s === null || s === undefined ? '' : String(s).replace(/\r?\n/g, ' ').replace(/\|/g, '\\|');
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
// The job summary: one row per REQUESTED model, in the order requested (AC-4).
|
|
317
|
+
function summaryMarkdown(d) {
|
|
318
|
+
const L = [];
|
|
319
|
+
L.push('### Driftproof');
|
|
320
|
+
L.push('');
|
|
321
|
+
L.push('| Model | Decision | Lift | Receipt |');
|
|
322
|
+
L.push('|---|---|---|---|');
|
|
323
|
+
for (const row of d.rows) {
|
|
324
|
+
const r = RENDER[row.state];
|
|
325
|
+
const delta = row.delta === null ? 'n/a' : (row.delta >= 0 ? '+' : '') + row.delta.toFixed(3);
|
|
326
|
+
const note = row.reason ? ` <br><sub>${cell(row.reason)}</sub>` : '';
|
|
327
|
+
L.push(`| \`${cell(row.model)}\` | ${r.marker} ${row.state} | ${delta} | ${cell(row.file) || '\u2014'}${note} |`);
|
|
328
|
+
}
|
|
329
|
+
L.push('');
|
|
330
|
+
L.push(`Decided over ${d.receiptCount} receipt(s) for ${d.requestedCount} requested model(s). `
|
|
331
|
+
+ `Worst: **${d.worst || 'none'}**.`);
|
|
332
|
+
if (d.unexpected.length) {
|
|
333
|
+
L.push('');
|
|
334
|
+
L.push(`Receipts in the directory that no requested model claimed: ${d.unexpected.join(', ')}.`);
|
|
335
|
+
}
|
|
336
|
+
return L.join('\n');
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
// The check title and the human line the enforcement step prints.
|
|
340
|
+
//
|
|
341
|
+
// The message names THE MODELS THAT REGRESSED, never the requested list. The
|
|
342
|
+
// old line interpolated the requested list, so one model's regression was
|
|
343
|
+
// reported against every model asked for - the same defect as the selection,
|
|
344
|
+
// seen from the other side.
|
|
345
|
+
// The parser's message still reaches the ANNOTATION, where the detail is worth
|
|
346
|
+
// having - but a workflow command is ONE LINE: a newline inside it ends the
|
|
347
|
+
// command and drops everything after, and an unbounded echo of a file's bytes
|
|
348
|
+
// does not belong in a check annotation either. One line, bounded, cut marked.
|
|
349
|
+
function oneLine(s, max = 160) {
|
|
350
|
+
const flat = String(s).replace(/\s+/g, ' ').trim();
|
|
351
|
+
return flat.length > max ? `${flat.slice(0, max - 1)}\u2026` : flat;
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
function enforcementLines(d, { failOnRegression = true } = {}) {
|
|
355
|
+
const L = [];
|
|
356
|
+
const per = d.rows.map((r) => `${r.model}: ${r.state}`).join(', ');
|
|
357
|
+
L.push(`Driftproof decision over ${d.receiptCount} receipt(s) for ${d.requestedCount} model(s) - ${per}`);
|
|
358
|
+
// AC-2: the count is reported before any verdict, and names each model that
|
|
359
|
+
// reached no readable receipt. Absence is not a pass.
|
|
360
|
+
//
|
|
361
|
+
// The two causes are reported apart (absence-vs-unreadable). A model with NO
|
|
362
|
+
// receipt and a model whose receipt exists and cannot be parsed both fail
|
|
363
|
+
// closed, but they send the reader to different places - one to the run that
|
|
364
|
+
// never wrote it, one to the file that is there - so a single message naming
|
|
365
|
+
// both as "no receipt was produced" would be false about one of them.
|
|
366
|
+
if (d.absent.length) {
|
|
367
|
+
L.push(`::error title=Driftproof::No receipt was produced for ${d.absent.join(', ')} - `
|
|
368
|
+
+ `${d.receiptCount} receipt(s) for ${d.requestedCount} requested model(s). `
|
|
369
|
+
+ `Failing closed: a model that was not measured did not pass.`);
|
|
370
|
+
}
|
|
371
|
+
if (d.unreadable.length) {
|
|
372
|
+
L.push(`::error title=Driftproof::The receipt for `
|
|
373
|
+
+ `${d.unreadable.map((u) => `${u.model} exists and is unreadable (${u.file}: ${oneLine(u.error)})`).join('; ')}. `
|
|
374
|
+
+ `Failing closed: a receipt that cannot be read is not a pass, and is not the same as one that was never written.`);
|
|
375
|
+
}
|
|
376
|
+
if (d.regressed.length) {
|
|
377
|
+
const deltas = d.rows.filter((r) => r.state === 'regression')
|
|
378
|
+
.map((r) => `${r.model} (delta ${r.delta})`).join(', ');
|
|
379
|
+
if (failOnRegression) {
|
|
380
|
+
L.push(`::error title=Driftproof::Skill REGRESSED on ${deltas}`);
|
|
381
|
+
} else {
|
|
382
|
+
L.push(`::warning title=Driftproof::Skill REGRESSED on ${deltas} (fail-on-regression is false)`);
|
|
383
|
+
}
|
|
384
|
+
}
|
|
385
|
+
// AC-4: an unmeasured run renders as unmeasured, never as passing, and does
|
|
386
|
+
// not fail the job on its own.
|
|
387
|
+
const unmeasured = d.rows.filter((r) => NEVER_SUCCESS.includes(r.state));
|
|
388
|
+
if (unmeasured.length) {
|
|
389
|
+
// AC-4 binds the TITLE: a `::warning` "whose title names the state and the
|
|
390
|
+
// models it came from". Until this it named `unmeasured[0].state` and no
|
|
391
|
+
// model at all - both facts were in the MESSAGE, which is not where the
|
|
392
|
+
// criterion puts them, and with two never-success states present the title
|
|
393
|
+
// carried only the first row's. A reader who sees the annotation collapsed
|
|
394
|
+
// to its title in a check list saw neither the second state nor any model
|
|
395
|
+
// (F-4 of 2026-09-12).
|
|
396
|
+
//
|
|
397
|
+
// Grouped by state, worst first, from STATE_ORDER - the same single ordering
|
|
398
|
+
// AC-1's enforcement and AC-3's badge read, so the title cannot disagree with
|
|
399
|
+
// them about which state governs.
|
|
400
|
+
//
|
|
401
|
+
// NO COMMA IN THE TITLE, and that is load-bearing rather than a style
|
|
402
|
+
// choice: GitHub parses a workflow command's properties as a
|
|
403
|
+
// COMMA-SEPARATED list, so `title=not measured,inconclusive` ends the title
|
|
404
|
+
// at the comma and leaves the rest to be read as a property GitHub does not
|
|
405
|
+
// know. States are joined with ' / ' and the models under one state with
|
|
406
|
+
// ' + ', neither of which GitHub reads.
|
|
407
|
+
const states = STATE_ORDER.filter((s) => unmeasured.some((r) => r.state === s));
|
|
408
|
+
const title = states
|
|
409
|
+
.map((s) => `${s} on ${unmeasured.filter((r) => r.state === s).map((r) => r.model).join(' + ')}`)
|
|
410
|
+
.join(' / ');
|
|
411
|
+
L.push(`::warning title=Driftproof: ${title}::`
|
|
412
|
+
+ `${unmeasured.map((r) => `${r.model}: ${r.state}`).join(', ')} - `
|
|
413
|
+
+ `this run did not measure these models, and is not a pass on them.`);
|
|
414
|
+
}
|
|
415
|
+
return L;
|
|
416
|
+
}
|
|
417
|
+
|
|
418
|
+
module.exports.RENDER = RENDER;
|
|
419
|
+
module.exports.rendersAsSuccess = rendersAsSuccess;
|
|
420
|
+
module.exports.badgeMessage = badgeMessage;
|
|
421
|
+
module.exports.badgeEndpointForSet = badgeEndpointForSet;
|
|
422
|
+
module.exports.githubOutputLines = githubOutputLines;
|
|
423
|
+
module.exports.summaryMarkdown = summaryMarkdown;
|
|
424
|
+
module.exports.enforcementLines = enforcementLines;
|
|
425
|
+
module.exports.VERDICT_WORD = VERDICT_WORD;
|
package/lib/runner.js
CHANGED
|
@@ -39,13 +39,38 @@ class Gate {
|
|
|
39
39
|
return rec.pass;
|
|
40
40
|
}
|
|
41
41
|
|
|
42
|
+
// Record an assertion whose SUBJECT this checkout does not carry (spec 030
|
|
43
|
+
// AC-5). Not a pass and not a failure: a row that says the check could not
|
|
44
|
+
// apply here and why.
|
|
45
|
+
//
|
|
46
|
+
// This exists because the two wrong answers are both worse. Guarding such a
|
|
47
|
+
// check with `if (...)` REMOVES the row, and the headline count silently
|
|
48
|
+
// drops by one - the register's `absence-vs-unreadable` class, and the reason
|
|
49
|
+
// the DECISIONS #4 check was written to fail rather than skip. Failing it
|
|
50
|
+
// instead means a `--depth 1` clone, which carries no `main` ref and so has
|
|
51
|
+
// no merge range to read, can never reach a green gate however correct it is.
|
|
52
|
+
// A third state costs one row and keeps both properties: the assertion always
|
|
53
|
+
// registers, and its cause is printed rather than hidden in a detail field
|
|
54
|
+
// that only prints on failure.
|
|
55
|
+
//
|
|
56
|
+
// `notApplicable` is NEVER the answer to "I could not read it". An absence
|
|
57
|
+
// must be positively detected - the ref is not there - or it is a failure.
|
|
58
|
+
notApplicable(name, cause) {
|
|
59
|
+
const rec = { section: this._section, name, pass: true, notApplicable: true, cause, detail: null };
|
|
60
|
+
this.results.push(rec);
|
|
61
|
+
// eslint-disable-next-line no-console
|
|
62
|
+
console.log(` [N/A] ${name} -- ${cause}`);
|
|
63
|
+
return true;
|
|
64
|
+
}
|
|
65
|
+
|
|
42
66
|
// Convenience: assert deep equality of two JSON-able values.
|
|
43
67
|
checkEqual(name, actual, expected) {
|
|
44
68
|
const pass = safeJson(actual) === safeJson(expected);
|
|
45
69
|
return this.check(name, pass, { actual, expected });
|
|
46
70
|
}
|
|
47
71
|
|
|
48
|
-
get passed() { return this.results.filter((r) => r.pass).length; }
|
|
72
|
+
get passed() { return this.results.filter((r) => r.pass && !r.notApplicable).length; }
|
|
73
|
+
get notApplicableCount() { return this.results.filter((r) => r.notApplicable).length; }
|
|
49
74
|
get failed() { return this.results.filter((r) => !r.pass); }
|
|
50
75
|
get total() { return this.results.length; }
|
|
51
76
|
|
|
@@ -53,12 +78,19 @@ class Gate {
|
|
|
53
78
|
summarize() {
|
|
54
79
|
const failed = this.failed;
|
|
55
80
|
// eslint-disable-next-line no-console
|
|
56
|
-
|
|
81
|
+
const na = this.notApplicableCount;
|
|
82
|
+
const naSuffix = na ? `, ${na} not applicable` : '';
|
|
83
|
+
// eslint-disable-next-line no-console
|
|
84
|
+
console.log(`\n=== ${this.title.toUpperCase()} RESULT: ${this.passed}/${this.total - na} passed, ${failed.length} failed${naSuffix} ===`);
|
|
85
|
+
for (const r of this.results.filter((x) => x.notApplicable)) {
|
|
86
|
+
// eslint-disable-next-line no-console
|
|
87
|
+
console.log(` N/A [${r.section}] ${r.name}: ${r.cause}`);
|
|
88
|
+
}
|
|
57
89
|
for (const f of failed) {
|
|
58
90
|
// eslint-disable-next-line no-console
|
|
59
91
|
console.log(` FAIL [${f.section}] ${f.name}: ${safeJson(f.detail)}`);
|
|
60
92
|
}
|
|
61
|
-
return { title: this.title, total: this.total, passed: this.passed, failed: failed.length, results: this.results };
|
|
93
|
+
return { title: this.title, total: this.total, passed: this.passed, failed: failed.length, notApplicable: na, results: this.results };
|
|
62
94
|
}
|
|
63
95
|
|
|
64
96
|
toExitCode() {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "driftproof",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.10.0",
|
|
4
4
|
"description": "A dated proof that this skill, this hash, this model, still helps: run a skill's eval suite with and without the skill across model versions, emit hash-verified dated receipts, and diff receipts into drift reports.",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
6
|
"keywords": [
|