driftproof 0.10.2 → 0.11.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -8
- package/bin/driftproof +22 -4
- package/config.js +10 -3
- package/lib/badge-svg.js +130 -0
- package/lib/decision.js +25 -8
- package/lib/diff.js +31 -2
- package/lib/hygiene.js +67 -2
- package/lib/receipt.js +5 -1
- package/lib/verdict.js +197 -26
- package/package.json +1 -1
- package/spec/RECEIPT.md +49 -10
- package/spec/receipt.schema.json +2 -2
- package/spec/receipt.v0.6.schema.json +1563 -0
package/README.md
CHANGED
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
<!-- SPDX-License-Identifier: Apache-2.0 -->
|
|
2
2
|
# Driftproof
|
|
3
3
|
|
|
4
|
-
**
|
|
4
|
+
**Is the gap your skill makes real, or noise? Did it hold on the last model release?**
|
|
5
5
|
|
|
6
|
-
A SKILL.md teaches an AI coding agent how you like things done.
|
|
7
|
-
|
|
8
|
-
skill
|
|
9
|
-
|
|
6
|
+
A SKILL.md teaches an AI coding agent how you like things done. A skill's own
|
|
7
|
+
tests passing is one answer. Driftproof asks two more: whether the scores with
|
|
8
|
+
the skill and without it separate beyond their spread, and whether that held
|
|
9
|
+
when the model changed. Each answer is a dated, hash-verified receipt, and when
|
|
10
|
+
there were too few draws to tell, the receipt says so.
|
|
10
11
|
|
|
11
12
|
[](https://driftproofhq.com)
|
|
12
13
|
— live badge for the bundled `commit-message-conventions` example, generated from its own receipt.
|
|
@@ -295,7 +296,7 @@ A receipt is the unit of evidence — one JSON document conforming to
|
|
|
295
296
|
|
|
296
297
|
```jsonc
|
|
297
298
|
{
|
|
298
|
-
"schema_version": "0.
|
|
299
|
+
"schema_version": "0.7",
|
|
299
300
|
"skill": { "name": "commit-message-conventions", "version": "0.2.0",
|
|
300
301
|
"content_hash": "…sha256 over SKILL.md + bundled files…" },
|
|
301
302
|
"suite": { "format": "agentskills.io/evals", "suite_hash": "…", "case_count": 10 },
|
|
@@ -304,7 +305,7 @@ A receipt is the unit of evidence — one JSON document conforming to
|
|
|
304
305
|
"model_release_date": "2025-10-01",
|
|
305
306
|
"provider": "anthropic",
|
|
306
307
|
"surface": "claude-cli",
|
|
307
|
-
"runner_version": "0.
|
|
308
|
+
"runner_version": "0.11.1",
|
|
308
309
|
"date_utc": "2026-07-27T…Z",
|
|
309
310
|
"registry": "registered",
|
|
310
311
|
"transcripts": "hashes-only",
|
|
@@ -383,7 +384,7 @@ jobs:
|
|
|
383
384
|
runs-on: ubuntu-latest
|
|
384
385
|
steps:
|
|
385
386
|
- uses: actions/checkout@v4
|
|
386
|
-
- uses: driftproofhq/driftproof@v0.
|
|
387
|
+
- uses: driftproofhq/driftproof@v0.11.1
|
|
387
388
|
with:
|
|
388
389
|
skill-dir: skills/my-skill
|
|
389
390
|
models: claude-haiku-4-5
|
package/bin/driftproof
CHANGED
|
@@ -13,7 +13,7 @@ const { buildDriftReport, revisionPairProblem } = require('../lib/diff');
|
|
|
13
13
|
const { surfaceForModel, isSubscriptionSurface, resolveModel, evalUser } = require('../lib/provider');
|
|
14
14
|
const { estimateRunCostUSD, BudgetTracker } = require('../lib/cost');
|
|
15
15
|
const { registryStatus, assertRegistered, REGISTRY_PATH } = require('../lib/models');
|
|
16
|
-
const { verdictFromReceipt, badgeEndpoint, githubOutputLines } = require('../lib/verdict');
|
|
16
|
+
const { verdictFromReceipt, badgeEndpoint, githubOutputLines, drawsLine, UNDERPOWERED_LINE } = require('../lib/verdict');
|
|
17
17
|
const decision = require('../lib/decision');
|
|
18
18
|
const { scaffoldInit } = require('../lib/init');
|
|
19
19
|
|
|
@@ -75,7 +75,7 @@ function loadRc(skillDir) {
|
|
|
75
75
|
|
|
76
76
|
// Flags that never take a value, so `--trusted-skill <skill-dir>` keeps the dir
|
|
77
77
|
// positional instead of swallowing it as the flag's value.
|
|
78
|
-
const BOOLEAN_FLAGS = new Set(['trusted-skill']);
|
|
78
|
+
const BOOLEAN_FLAGS = new Set(['trusted-skill', 'svg']);
|
|
79
79
|
|
|
80
80
|
// ── THE INPUT CONTRACT (spec 026 AC-9, F5) ────────────────────────────────────
|
|
81
81
|
// A COPY of action/lib.sh's RE_* rules, in the shape spec 028's fixture holds
|
|
@@ -177,7 +177,7 @@ USAGE
|
|
|
177
177
|
[--keep-transcripts] [--out DIR] [--trusted-skill]
|
|
178
178
|
${PROJECT_NAME} diff <receiptA.json> <receiptB.json> [--out FILE] [--mode release|revision]
|
|
179
179
|
${PROJECT_NAME} validate <receipt.json>
|
|
180
|
-
${PROJECT_NAME} badge <receipt.json|receipts-dir> [--models a,b] [--out FILE] [--github-output]
|
|
180
|
+
${PROJECT_NAME} badge <receipt.json|receipts-dir> [--models a,b] [--out FILE] [--github-output] [--svg [--href URL]]
|
|
181
181
|
${PROJECT_NAME} decide <receipts-dir> --models a,b [--github-output] [--badge FILE]
|
|
182
182
|
[--summary FILE] [--enforce] [--fail-on-regression true|false]
|
|
183
183
|
${PROJECT_NAME} import <results.json> --from agent-skills-eval|skillgrade [--out DIR]
|
|
@@ -515,17 +515,35 @@ function cmdBadge(positional, flags) {
|
|
|
515
515
|
if (fs.existsSync(p) && fs.statSync(p).isDirectory()) return cmdBadgeSet(p, flags);
|
|
516
516
|
const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
517
517
|
refuseUnverified(receipt, p, 'badge');
|
|
518
|
+
if (flags.svg) {
|
|
519
|
+
// SPEC 036. The drawn badge: state, model, date, lift and uncertainty, with the
|
|
520
|
+
// machine token in its data attributes. After the same hash check as the JSON.
|
|
521
|
+
const svg = require('../lib/badge-svg').badgeSvg(receipt, { href: typeof flags.href === 'string' ? flags.href : null });
|
|
522
|
+
if (flags.out) {
|
|
523
|
+
const out = path.resolve(flags.out);
|
|
524
|
+
fs.mkdirSync(path.dirname(out), { recursive: true });
|
|
525
|
+
fs.writeFileSync(out, svg);
|
|
526
|
+
console.log(`badge written to ${flags.out} (${verdictFromReceipt(receipt).verdict}, svg)`);
|
|
527
|
+
} else process.stdout.write(svg);
|
|
528
|
+
return;
|
|
529
|
+
}
|
|
518
530
|
if (flags['github-output']) { console.log(githubOutputLines(receipt)); return; }
|
|
519
531
|
const badge = badgeEndpoint(receipt);
|
|
520
532
|
const json = JSON.stringify(badge, null, 2);
|
|
533
|
+
const v = verdictFromReceipt(receipt);
|
|
534
|
+
// SPEC 035 AC-3. An UNDERPOWERED receipt says so in words and says what it would
|
|
535
|
+
// have needed. Beside JSON on stdout the two lines go to stderr, so a caller that
|
|
536
|
+
// redirects the badge still gets a file that parses.
|
|
537
|
+
const power = v.verdict === 'UNDERPOWERED' ? [`${UNDERPOWERED_LINE}.`, drawsLine(v.drawsNeeded)] : [];
|
|
521
538
|
if (flags.out) {
|
|
522
539
|
const out = path.resolve(flags.out);
|
|
523
540
|
fs.mkdirSync(path.dirname(out), { recursive: true });
|
|
524
541
|
fs.writeFileSync(out, json + '\n');
|
|
525
|
-
const v = verdictFromReceipt(receipt);
|
|
526
542
|
console.log(`badge written to ${flags.out} (${v.verdict}: ${badge.message}, ${badge.color})`);
|
|
543
|
+
for (const line of power) console.log(line);
|
|
527
544
|
} else {
|
|
528
545
|
console.log(json);
|
|
546
|
+
for (const line of power) console.error(line);
|
|
529
547
|
}
|
|
530
548
|
}
|
|
531
549
|
|
package/config.js
CHANGED
|
@@ -9,7 +9,7 @@ const PROJECT_NAME = 'driftproof';
|
|
|
9
9
|
// Bumped whenever the runner's behaviour or receipt-generation semantics change
|
|
10
10
|
// in a way that could affect results. Recorded into every receipt as
|
|
11
11
|
// run.runner_version so a receipt is reproducible against a known engine.
|
|
12
|
-
const RUNNER_VERSION = '0.
|
|
12
|
+
const RUNNER_VERSION = '0.11.1';
|
|
13
13
|
|
|
14
14
|
// The eval format we CONSUME (we deliberately do not invent our own).
|
|
15
15
|
const SUITE_FORMAT = 'agentskills.io/evals';
|
|
@@ -38,7 +38,7 @@ const SUITE_FORMAT = 'agentskills.io/evals';
|
|
|
38
38
|
// results.aggregates.band_rule, and bands that are null where the formula
|
|
39
39
|
// cannot form. NOT additive for the validator: a v0.5 receipt restamped 0.6
|
|
40
40
|
// is refused. v0.5 is frozen as receipt.v0.5.schema.json.
|
|
41
|
-
const RECEIPT_SCHEMA_VERSION = '0.
|
|
41
|
+
const RECEIPT_SCHEMA_VERSION = '0.7';
|
|
42
42
|
|
|
43
43
|
// Input bounds on the skill loader and the post-checks (spec 026 AC-13, audit
|
|
44
44
|
// A3/A7). lib/skill.js loaded every bundled file with no count, depth, byte,
|
|
@@ -104,6 +104,13 @@ const DEFAULT_JUDGE_SAMPLES = 5;
|
|
|
104
104
|
// spec/RECEIPT.md § "Drift verdict rule" and the report methodology.
|
|
105
105
|
const EFFECT_FLOOR = 0.05;
|
|
106
106
|
|
|
107
|
+
// How many standard errors of the difference between two means a case's resolution
|
|
108
|
+
// adds to its two spreads (spec 035 R-2). A case the band rule does not separate is
|
|
109
|
+
// UNDERPOWERED when its spreads plus this many standard errors reach the floor: at
|
|
110
|
+
// those spreads and draws, a true shift of the floor's size could have gone unseparated.
|
|
111
|
+
// Named, so a later decision moves it by one line and one amendment.
|
|
112
|
+
const POWER_Z = 2;
|
|
113
|
+
|
|
107
114
|
// ── generation sampling (receipt spec v0.5) ─────────────────────────────────
|
|
108
115
|
//
|
|
109
116
|
// Report #006 measured across-draw spread at sd 0.186 and 0.183 while the
|
|
@@ -157,7 +164,7 @@ const REPORT_005_JUDGE_MODEL = 'claude-haiku-4-5';
|
|
|
157
164
|
module.exports = {
|
|
158
165
|
PROJECT_NAME, RUNNER_VERSION, SUITE_FORMAT, RECEIPT_SCHEMA_VERSION, DEFAULT_JUDGE_SAMPLES,
|
|
159
166
|
SKILL_MAX_FILES, SKILL_MAX_BYTES, SKILL_MAX_DEPTH, SUITE_MAX_CASES, CASE_MAX_CHARS, CHECK_MAX_PATTERN,
|
|
160
|
-
EFFECT_FLOOR, DEV_MAX_USD, DEV_MAX_CALLS, REPORT_MAX_USD, TRIGGER_MAX_USD,
|
|
167
|
+
EFFECT_FLOOR, POWER_Z, DEV_MAX_USD, DEV_MAX_CALLS, REPORT_MAX_USD, TRIGGER_MAX_USD,
|
|
161
168
|
GENERATION_SAMPLES_MIN, GENERATION_SAMPLES_MAX, GENERATION_SD_THRESHOLD, GENERATION_STABILITY_EPS,
|
|
162
169
|
REPORT_002_CLAUDE_MODEL, REPORT_002_GPT_MODEL, REPORT_002_JUDGE_MODEL,
|
|
163
170
|
REPORT_003_NEW_MODEL, REPORT_003_OLD_MODEL, REPORT_003_JUDGE_MODEL,
|
package/lib/badge-svg.js
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// The badge, drawn (spec 036).
|
|
5
|
+
//
|
|
6
|
+
// The shields endpoint JSON (lib/verdict.js badgeEndpoint) carries a label, a message
|
|
7
|
+
// and a colour, which is all a shields badge can hold. This draws the badge itself, so
|
|
8
|
+
// it can show what a reader needs at a glance: the state, the model, the date, the lift
|
|
9
|
+
// and its uncertainty. Four segments, left to right:
|
|
10
|
+
//
|
|
11
|
+
// driftproof | <state label> | <model> · <date> | lift <+0.000> <± 0.000 | no ±>
|
|
12
|
+
//
|
|
13
|
+
// The fourth segment carries the lift ONLY for a state that measured one. UNDERPOWERED and
|
|
14
|
+
// NOT_MEASURED carry the draws taken against the draws needed instead (A-036-2):
|
|
15
|
+
//
|
|
16
|
+
// driftproof | not enough draws | <model> · <date> | 3 draws of 87 needed
|
|
17
|
+
//
|
|
18
|
+
// THE WORDS ARE HUMAN; THE TOKENS ARE DATA. What a reader sees is a label (passing, not
|
|
19
|
+
// enough draws); the machine token the Action and the differ publish sits on the root
|
|
20
|
+
// element as data-verdict, with the receipt hash beside it, and never in the text.
|
|
21
|
+
//
|
|
22
|
+
// UNDERPOWERED IS DRAWN AS THE HONEST STATE, not as a failure: a calm indigo no other
|
|
23
|
+
// state uses, an open hatch over it, a dashed edge around it, and the words "not enough
|
|
24
|
+
// draws". Every state's text is white on a fill it clears 4.5:1 against, hatch included.
|
|
25
|
+
//
|
|
26
|
+
// Text widths are fixed with textLength, so a text's rendered box is the width this file
|
|
27
|
+
// allots it whatever font the viewer has, and it cannot spill out of its segment.
|
|
28
|
+
|
|
29
|
+
const { receiptVerdict, shortModel, receiptDrawsTaken } = require('./verdict');
|
|
30
|
+
|
|
31
|
+
const INK = '#ffffff';
|
|
32
|
+
const BRAND = '#24292f';
|
|
33
|
+
const DETAIL = '#4b5563';
|
|
34
|
+
const EFFECT = '#374151';
|
|
35
|
+
const STATE = {
|
|
36
|
+
PASSED: { fill: '#1a7f37' },
|
|
37
|
+
REGRESSED: { fill: '#b42318' },
|
|
38
|
+
UNDERPOWERED: { fill: '#3538cd', hatch: '#444ce7', edge: '#a4bcfd' },
|
|
39
|
+
NO_EFFECT: { fill: '#57606a' },
|
|
40
|
+
NOT_MEASURED: { fill: '#656d76' },
|
|
41
|
+
};
|
|
42
|
+
const LABELS = {
|
|
43
|
+
PASSED: 'passing',
|
|
44
|
+
REGRESSED: 'regressed',
|
|
45
|
+
UNDERPOWERED: 'not enough draws',
|
|
46
|
+
NO_EFFECT: 'no separation detected',
|
|
47
|
+
NOT_MEASURED: 'not measured',
|
|
48
|
+
};
|
|
49
|
+
|
|
50
|
+
const H = 20;
|
|
51
|
+
const PAD = 7;
|
|
52
|
+
const CHAR = 6.6;
|
|
53
|
+
const esc = (s) => String(s).replace(/&/g, '&').replace(/</g, '<').replace(/>/g, '>').replace(/"/g, '"');
|
|
54
|
+
const textW = (s) => Math.round(String(s).length * CHAR * 10) / 10;
|
|
55
|
+
|
|
56
|
+
// A LIFT FIGURE IS A MEASUREMENT, so only a state that measured one may print it
|
|
57
|
+
// (spec 036 A-036-2). UNDERPOWERED says the receipt's draws cannot resolve a shift of the
|
|
58
|
+
// effect floor, and NOT_MEASURED says the receipt carries no verdict at all. A lift beside
|
|
59
|
+
// either is the badge answering, in its most legible segment, a question the state has just
|
|
60
|
+
// said it cannot answer -- and a reader who takes one figure from a badge takes that one.
|
|
61
|
+
// Those two states print what the reader can act on instead: the draws taken against the
|
|
62
|
+
// draws that would have been needed.
|
|
63
|
+
function effectText(receipt, v) {
|
|
64
|
+
const c = receipt.comparison || {};
|
|
65
|
+
if (v.verdict === 'UNDERPOWERED' || v.verdict === 'NOT_MEASURED') return drawsText(receipt, v);
|
|
66
|
+
const lift = typeof c.delta === 'number' ? `lift ${c.delta >= 0 ? '+' : '-'}${Math.abs(c.delta).toFixed(3)}` : 'no lift';
|
|
67
|
+
const unc = typeof c.delta_uncertainty === 'number' ? `± ${c.delta_uncertainty.toFixed(3)}`
|
|
68
|
+
: c.delta_uncertainty_unavailable === 'single_case' ? 'no ± (1 case)' : 'no ±';
|
|
69
|
+
return `${lift} ${unc}`;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
// Draws taken against draws needed, in the badge's words. The needed count is
|
|
73
|
+
// receiptDrawsNeeded's, the same figure drawsLine prints in a sentence, so the badge and
|
|
74
|
+
// the page cannot say two different numbers. Where no count exists the badge says which
|
|
75
|
+
// of the two reasons it is, rather than printing a bare figure or nothing at all.
|
|
76
|
+
//
|
|
77
|
+
// BOTH NUMBERS ARE ONE CASE'S (A-036-7). Where a case sets the draws needed, the draws taken
|
|
78
|
+
// are that case's own, which receiptDrawsNeeded carries as `taken`. This read the smallest
|
|
79
|
+
// over every readable case, so a receipt whose binding case drew 4 and whose other case drew
|
|
80
|
+
// 2 badged "2 draws of 5 needed": a 2 from a case the 5 does not belong to. Where no case sets
|
|
81
|
+
// the draws needed (NOT_MEASURED carries no verdict to compute one from) there is no binding
|
|
82
|
+
// case, and the smallest over every readable case is the figure.
|
|
83
|
+
function drawsText(receipt, v) {
|
|
84
|
+
const d = v.drawsNeeded;
|
|
85
|
+
const taken = d ? d.taken : receiptDrawsTaken(receipt);
|
|
86
|
+
if (taken === null || taken === undefined) return 'no draws recorded';
|
|
87
|
+
const draws = `${taken} ${taken === 1 ? 'draw' : 'draws'}`;
|
|
88
|
+
if (!d) return `${draws}, needed not computed`;
|
|
89
|
+
if (typeof d.value === 'number') return `${draws} of ${d.value} needed`;
|
|
90
|
+
return `${draws}, no count at these spreads`;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
function badgeSvg(receipt, { href = null } = {}) {
|
|
94
|
+
const v = receiptVerdict(receipt);
|
|
95
|
+
const state = STATE[v.verdict];
|
|
96
|
+
const label = LABELS[v.verdict];
|
|
97
|
+
const model = shortModel((receipt.run || {}).model_id);
|
|
98
|
+
const date = String((receipt.run || {}).date_utc || '').slice(0, 10);
|
|
99
|
+
const segs = [
|
|
100
|
+
{ seg: 'brand', text: 'driftproof', fill: BRAND },
|
|
101
|
+
{ seg: 'state', text: label, fill: state.fill },
|
|
102
|
+
{ seg: 'detail', text: `${model} · ${date}`, fill: DETAIL },
|
|
103
|
+
{ seg: 'effect', text: effectText(receipt, v), fill: EFFECT },
|
|
104
|
+
];
|
|
105
|
+
let x = 0;
|
|
106
|
+
for (const s of segs) { s.w = Math.round(textW(s.text) + 2 * PAD); s.x = x; x += s.w; }
|
|
107
|
+
const width = x;
|
|
108
|
+
const title = `driftproof: ${label} on ${model}, ${date}, ${effectText(receipt, v)}`;
|
|
109
|
+
const L = [];
|
|
110
|
+
L.push(`<svg xmlns="http://www.w3.org/2000/svg" width="${width}" height="${H}" viewBox="0 0 ${width} ${H}" role="img" aria-label="${esc(title)}" data-verdict="${v.verdict}" data-receipt-hash="${esc(receipt.receipt_hash || '')}">`);
|
|
111
|
+
L.push(`<title>${esc(title)}</title>`);
|
|
112
|
+
if (state.hatch) {
|
|
113
|
+
L.push(`<defs><pattern id="hatch" width="6" height="6" patternUnits="userSpaceOnUse" patternTransform="rotate(45)"><line x1="0" y1="0" x2="0" y2="6" stroke="${state.hatch}" stroke-width="3"/></pattern></defs>`);
|
|
114
|
+
}
|
|
115
|
+
if (href) L.push(`<a href="${esc(href)}">`);
|
|
116
|
+
for (const s of segs) {
|
|
117
|
+
const isState = s.seg === 'state';
|
|
118
|
+
const edge = isState && state.edge ? ` stroke="${state.edge}" stroke-width="1" stroke-dasharray="3 2"` : '';
|
|
119
|
+
L.push(`<rect data-seg="${s.seg}" x="${s.x}" y="0" width="${s.w}" height="${H}" fill="${s.fill}"${edge}/>`);
|
|
120
|
+
if (isState && state.hatch) L.push(`<rect data-seg="state-hatch" x="${s.x}" y="0" width="${s.w}" height="${H}" fill="url(#hatch)" opacity="0.35"/>`);
|
|
121
|
+
}
|
|
122
|
+
for (const s of segs) {
|
|
123
|
+
L.push(`<text data-seg="${s.seg}" x="${s.x + PAD}" y="14" fill="${INK}" font-family="Verdana,DejaVu Sans,sans-serif" font-size="11" textLength="${textW(s.text)}" lengthAdjust="spacingAndGlyphs">${esc(s.text)}</text>`);
|
|
124
|
+
}
|
|
125
|
+
if (href) L.push('</a>');
|
|
126
|
+
L.push('</svg>');
|
|
127
|
+
return L.join('\n') + '\n';
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
module.exports = { badgeSvg, LABELS, STATE };
|
package/lib/decision.js
CHANGED
|
@@ -20,7 +20,7 @@
|
|
|
20
20
|
const fs = require('fs');
|
|
21
21
|
const path = require('path');
|
|
22
22
|
const { EFFECT_FLOOR } = require('../config');
|
|
23
|
-
const { shortModel, githubOutputEntry } = require('./verdict');
|
|
23
|
+
const { shortModel, githubOutputEntry, receiptVerdict, drawsLine, UNDERPOWERED_LINE } = require('./verdict');
|
|
24
24
|
|
|
25
25
|
// The six decision states (spec 030 AC-4).
|
|
26
26
|
//
|
|
@@ -29,7 +29,12 @@ const { shortModel, githubOutputEntry } = require('./verdict');
|
|
|
29
29
|
// that says so. `no detected effect` is false about such a model - the effect
|
|
30
30
|
// was detected and it cleared the floor - and widening it to mean "not a
|
|
31
31
|
// regression" would put a measured lift and a measured nothing in one cell.
|
|
32
|
-
|
|
32
|
+
//
|
|
33
|
+
// `underpowered` is the seventh (spec 035): the receipt measured, and its own spreads
|
|
34
|
+
// and draws could not resolve a shift of the effect floor. It sits after `not
|
|
35
|
+
// measured`, which measured nothing, and before `no detected effect`, which measured
|
|
36
|
+
// with a resolution below the floor.
|
|
37
|
+
const STATES = ['regression', 'refused', 'inconclusive', 'not measured', 'underpowered', 'no detected effect', 'helped'];
|
|
33
38
|
|
|
34
39
|
// WORST FIRST. This is the single definition of the ordering; AC-1's
|
|
35
40
|
// enforcement and AC-3's badge both read it, so they cannot disagree about
|
|
@@ -38,7 +43,7 @@ const STATE_ORDER = STATES.slice();
|
|
|
38
43
|
|
|
39
44
|
// States that must never render as success on any surface the action writes -
|
|
40
45
|
// the badge, the summary row, or the check title (AC-4).
|
|
41
|
-
const NEVER_SUCCESS = ['inconclusive', 'not measured'];
|
|
46
|
+
const NEVER_SUCCESS = ['inconclusive', 'not measured', 'underpowered'];
|
|
42
47
|
|
|
43
48
|
// States that fail the job. `refused` fails CLOSED (AC-2): a model that
|
|
44
49
|
// produced no receipt is not a model that passed. `regression` fails subject to
|
|
@@ -86,11 +91,15 @@ function decisionState(receipt) {
|
|
|
86
91
|
if (level !== 'TESTED' || kind !== 'model') return 'not measured';
|
|
87
92
|
const cmp = receipt.comparison || {};
|
|
88
93
|
if ((receipt.run && receipt.run.status === 'incomplete') || typeof cmp.delta !== 'number') return 'inconclusive';
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
94
|
+
// Spec 035: the measured states are the receipt verdict's, read per case by the
|
|
95
|
+
// band rule (lib/verdict.js receiptVerdict), not the aggregate delta against the floor.
|
|
96
|
+
// receiptVerdict refuses on the same four conditions the two clauses above split
|
|
97
|
+
// between `not measured` and `inconclusive`, so it cannot answer NOT_MEASURED here.
|
|
98
|
+
return MEASURED_STATE[receiptVerdict(receipt).verdict];
|
|
92
99
|
}
|
|
93
100
|
|
|
101
|
+
const MEASURED_STATE = { REGRESSED: 'regression', PASSED: 'helped', UNDERPOWERED: 'underpowered', NO_EFFECT: 'no detected effect' };
|
|
102
|
+
|
|
94
103
|
// The worst state in a set, by STATE_ORDER. An empty set has no decision, and
|
|
95
104
|
// says so with null rather than defaulting to something benign.
|
|
96
105
|
function worstState(states) {
|
|
@@ -182,6 +191,8 @@ function decideSet(dir, requestedCsv, { failOnRegression = true } = {}) {
|
|
|
182
191
|
delta: receipt && receipt.comparison && typeof receipt.comparison.delta === 'number'
|
|
183
192
|
? receipt.comparison.delta : null,
|
|
184
193
|
file: hit ? path.basename(hit.file) : (unreadable ? path.basename(unreadable.file) : null),
|
|
194
|
+
// Spec 035 AC-3: what an underpowered row would have needed, from the same reading.
|
|
195
|
+
drawsNeeded: state === 'underpowered' ? receiptVerdict(receipt).drawsNeeded : null,
|
|
185
196
|
// The distinction is carried on the ROW, not left in a sentence, because
|
|
186
197
|
// the enforcement message has to say which of the two happened (AC-2's
|
|
187
198
|
// mutation class, absence-vs-unreadable).
|
|
@@ -246,6 +257,7 @@ const RENDER = {
|
|
|
246
257
|
refused: { word: 'refused', color: 'red', marker: '\u274c' },
|
|
247
258
|
inconclusive: { word: 'inconclusive', color: 'yellow', marker: '\u26a0\ufe0f' },
|
|
248
259
|
'not measured': { word: 'not measured', color: 'lightgrey', marker: '\u26a0\ufe0f' },
|
|
260
|
+
underpowered: { word: 'not enough draws', color: 'blue', marker: '\u26a0\ufe0f' },
|
|
249
261
|
'no detected effect': { word: 'no effect', color: 'lightgrey', marker: '\u2014' },
|
|
250
262
|
helped: { word: 'passing', color: 'brightgreen', marker: '\u2705' },
|
|
251
263
|
};
|
|
@@ -283,7 +295,7 @@ function badgeEndpointForSet(d, { label = 'driftproof' } = {}) {
|
|
|
283
295
|
// `worst`, `regressed` and `missing` are new and carry what it could not say.
|
|
284
296
|
const VERDICT_WORD = {
|
|
285
297
|
regression: 'REGRESSED', refused: 'REFUSED', inconclusive: 'INCONCLUSIVE',
|
|
286
|
-
'not measured': 'NOT_MEASURED', 'no detected effect': 'NO_EFFECT', helped: 'PASSED',
|
|
298
|
+
'not measured': 'NOT_MEASURED', underpowered: 'UNDERPOWERED', 'no detected effect': 'NO_EFFECT', helped: 'PASSED',
|
|
287
299
|
};
|
|
288
300
|
|
|
289
301
|
function githubOutputLines(d) {
|
|
@@ -299,6 +311,10 @@ function githubOutputLines(d) {
|
|
|
299
311
|
['regressed_models', d.regressed.join(',')],
|
|
300
312
|
['missing_models', d.missing.join(',')],
|
|
301
313
|
['receipt_count', d.receiptCount],
|
|
314
|
+
// Spec 035 AC-3: the draws the worst row would have needed, when that row is
|
|
315
|
+
// underpowered; `none` when no draw count reaches the floor at its spreads, and
|
|
316
|
+
// empty otherwise.
|
|
317
|
+
['draws_needed', worstRow && worstRow.drawsNeeded ? (worstRow.drawsNeeded.value === null ? 'none' : worstRow.drawsNeeded.value) : ''],
|
|
302
318
|
].map(([k, v]) => githubOutputEntry(k, v)).join('\n');
|
|
303
319
|
}
|
|
304
320
|
|
|
@@ -324,7 +340,8 @@ function summaryMarkdown(d) {
|
|
|
324
340
|
const r = RENDER[row.state];
|
|
325
341
|
const delta = row.delta === null ? 'n/a' : (row.delta >= 0 ? '+' : '') + row.delta.toFixed(3);
|
|
326
342
|
const note = row.reason ? ` <br><sub>${cell(row.reason)}</sub>` : '';
|
|
327
|
-
|
|
343
|
+
const power = row.state === 'underpowered' ? ` <br><sub>${cell(UNDERPOWERED_LINE)}. ${cell(drawsLine(row.drawsNeeded))}</sub>` : '';
|
|
344
|
+
L.push(`| \`${cell(row.model)}\` | ${r.marker} ${row.state}${power} | ${delta} | ${cell(row.file) || '\u2014'}${note} |`);
|
|
328
345
|
}
|
|
329
346
|
L.push('');
|
|
330
347
|
L.push(`Decided over ${d.receiptCount} receipt(s) for ${d.requestedCount} requested model(s). `
|
package/lib/diff.js
CHANGED
|
@@ -5,6 +5,7 @@ const { bandVerdict, round, WITHIN_NOISE } = require('./stats');
|
|
|
5
5
|
const { EFFECT_FLOOR } = require('../config');
|
|
6
6
|
const { revisionHeadline } = require('./revision');
|
|
7
7
|
const { baselineReproduces, REFUSAL_REASONS, bandOf } = require('./reuse');
|
|
8
|
+
const { caseRule, armOf, drawsLine, UNDERPOWERED_LINE } = require('./verdict');
|
|
8
9
|
|
|
9
10
|
// Practical-significance gate applied ON TOP of band separation. bandVerdict()
|
|
10
11
|
// stays a pure geometry test (kept that way so its unit checks are unambiguous);
|
|
@@ -198,6 +199,15 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
198
199
|
// untested input does: no delta is asserted, and the reason travels with it.
|
|
199
200
|
const measured = belowTested.length === 0 && !refusal && incompleteSides.length === 0 && !judgeProblem;
|
|
200
201
|
|
|
202
|
+
// THE UNDERPOWERED STATE RIDES BESIDE THE VERDICT VALUE (spec 035 R-6). The value
|
|
203
|
+
// is unchanged, because Reports 006 to 008 and spec 009's gate compare against it
|
|
204
|
+
// and a published record quotes the lines it produced. `power` says, for a case
|
|
205
|
+
// the rule did not separate, whether the two with_skill arms' own spreads and
|
|
206
|
+
// draws could have resolved a shift of the effect floor, by the rule lib/verdict.js
|
|
207
|
+
// applies to a single receipt.
|
|
208
|
+
const rawWith = (r) => new Map(r.results.cases.filter((c) => c.mode === 'with_skill').map((c) => [c.id, c]));
|
|
209
|
+
const aRaw = rawWith(a);
|
|
210
|
+
const bRaw = rawWith(b);
|
|
201
211
|
const perCase = ids.map((id) => {
|
|
202
212
|
const before = aB[id] || null;
|
|
203
213
|
const after = bB[id] || null;
|
|
@@ -205,7 +215,13 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
205
215
|
const verdict = refusal ? 'refused'
|
|
206
216
|
: !measured ? 'not measured'
|
|
207
217
|
: (before && after) ? verdictWithFloor(before, after, delta) : 'n/a';
|
|
208
|
-
|
|
218
|
+
let power = null;
|
|
219
|
+
if (measured && before && after && (verdict === WITHIN_NOISE || verdict === WITHIN_NOISE_FLOOR)) {
|
|
220
|
+
const ra = armOf(aRaw.get(id)); const rb = armOf(bRaw.get(id));
|
|
221
|
+
const rule = ra && rb ? caseRule(ra, rb) : null;
|
|
222
|
+
if (rule && rule.state === 'underpowered') power = { state: 'underpowered', drawsNeeded: rule.drawsNeeded, reason: rule.reason, ...(rule.spread !== undefined ? { spread: rule.spread } : {}) };
|
|
223
|
+
}
|
|
224
|
+
return { id, before, after, delta, verdict, power };
|
|
209
225
|
});
|
|
210
226
|
// Sort worst-first: regressions, then by delta.
|
|
211
227
|
const order = { regression: 0, [WITHIN_NOISE]: 1, [WITHIN_NOISE_FLOOR]: 1, improvement: 2, 'n/a': 3, 'not measured': 3, refused: 3 };
|
|
@@ -391,6 +407,18 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
391
407
|
// the rule, which is what the headline above says.
|
|
392
408
|
L.push(`with_skill mean moved ${fmt(headlineDelta)} (${bandStr(aAgg)} → ${bandStr(bAgg)}; band = suite dispersion). Per-case band-overlap verdicts: ${regressions.length} regression(s), ${perCase.filter((r) => r.verdict === 'improvement').length} improvement(s), ${nWithin - nFloor} with no separation detected${nFloor ? `, ${nFloor} below the ${EFFECT_FLOOR} effect floor` : ''}.`);
|
|
393
409
|
L.push('');
|
|
410
|
+
const underpowered = perCase.filter((r) => r.power);
|
|
411
|
+
if (underpowered.length) {
|
|
412
|
+
// One count, one sentence, one draws line: the worst case's (spec 035 AC-5).
|
|
413
|
+
const spread = underpowered.find((r) => r.power.reason === 'spread');
|
|
414
|
+
const most = underpowered.filter((r) => r.power.reason === 'draws').sort((x, y) => y.power.drawsNeeded - x.power.drawsNeeded)[0];
|
|
415
|
+
const worst = spread ? { value: null, reason: 'spread', case: spread.id, spread: spread.power.spread }
|
|
416
|
+
: most ? { value: most.power.drawsNeeded, reason: 'draws', case: most.id }
|
|
417
|
+
: { value: 2, reason: 'single_draw', case: underpowered[0].id };
|
|
418
|
+
const powerLine = `Underpowered: ${underpowered.length} of the cases with no separation under the rule. ${UNDERPOWERED_LINE}. ${drawsLine(worst)}`;
|
|
419
|
+
L.push(powerLine);
|
|
420
|
+
L.push('');
|
|
421
|
+
}
|
|
394
422
|
}
|
|
395
423
|
|
|
396
424
|
L.push(`## Per-case with_skill (band overlap → verdict)`);
|
|
@@ -399,7 +427,8 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
399
427
|
L.push(`|---|---|---|---|---|`);
|
|
400
428
|
for (const r of perCase) {
|
|
401
429
|
const flag = r.verdict === 'regression' ? '🔻 regression' : r.verdict === 'improvement' ? '🔼 improvement' : r.verdict === WITHIN_NOISE_FLOOR ? 'below effect floor' : r.verdict === WITHIN_NOISE ? 'no separation detected' : r.verdict === 'not measured' ? 'not measured' : 'n/a';
|
|
402
|
-
|
|
430
|
+
const label = r.power ? `${flag}; ${UNDERPOWERED_LINE}` : flag;
|
|
431
|
+
L.push(`| \`${r.id}\` | ${r.before ? bandStr(r.before) : 'n/a'} | ${r.after ? bandStr(r.after) : 'n/a'} | ${fmt(r.delta)} | ${label} |`);
|
|
403
432
|
}
|
|
404
433
|
L.push('');
|
|
405
434
|
|
package/lib/hygiene.js
CHANGED
|
@@ -23,6 +23,8 @@
|
|
|
23
23
|
// Patterns are written so a regex literal cannot match its own text: a
|
|
24
24
|
// metacharacter or a character class follows each fixed prefix.
|
|
25
25
|
|
|
26
|
+
const zlib = require('zlib');
|
|
27
|
+
|
|
26
28
|
const dec = (b64) => JSON.parse(Buffer.from(b64, 'base64').toString('utf8'));
|
|
27
29
|
|
|
28
30
|
// Build-box host + private IP, base64 so the plaintext never lands in a file.
|
|
@@ -54,7 +56,70 @@ function scanPath(rel) {
|
|
|
54
56
|
return /(^|\/)\.env(\.|$)/.test(rel) ? [{ file: rel, kind: 'env-file' }] : [];
|
|
55
57
|
}
|
|
56
58
|
|
|
59
|
+
// A PNG'S TEXT-BEARING REGIONS (spec 038 A-038-5, the classification spec 027 already made for the
|
|
60
|
+
// repository gate's confidentiality scan, applied here).
|
|
61
|
+
//
|
|
62
|
+
// THE THREAT MODEL IS ACCIDENTAL PLAINTEXT DISCLOSURE by a person or an agent writing the tree
|
|
63
|
+
// (CONSTITUTION § Threat model). A PNG carries text in its tEXt, iTXt and zTXt chunks, and carries
|
|
64
|
+
// pixels in IDAT, which is deflate output: nobody types into it, and a run of bytes that happens to
|
|
65
|
+
// decode as an address there is a coincidence of compression, not a disclosure. Scanned as one
|
|
66
|
+
// string, a 2 MB screenshot is two million chances of that coincidence, and it has fired twice on
|
|
67
|
+
// this repository's own committed captures.
|
|
68
|
+
//
|
|
69
|
+
// So: every chunk but IDAT is scanned, chunk type names included, and IDAT is not. A name RENDERED
|
|
70
|
+
// INTO a screenshot is pixels, which no byte scan sees either way - that is the gap this scan has
|
|
71
|
+
// always had, and it is unchanged. Everything that is not a PNG is scanned whole, as before.
|
|
72
|
+
//
|
|
73
|
+
// FAIL CLOSED WHERE THE FILE STOPS BEING A PNG (spec 038 approval F-3). Bytes after IEND are no chunk
|
|
74
|
+
// at all, and a chunk whose length runs past the end of the file is a truncated file: either way the
|
|
75
|
+
// remainder, from where the walk lost the structure, is scanned whole, as every file was before this
|
|
76
|
+
// classification. And a text chunk is read in the encoding it carries text in: a zTXt chunk's text
|
|
77
|
+
// and a compressed iTXt chunk's text are inflated and scanned as well as their raw bytes. A chunk
|
|
78
|
+
// that does not inflate is scanned raw, which it already was.
|
|
79
|
+
const PNG_SIG = Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a]);
|
|
80
|
+
const INFLATE_CAP = 16 * 1024 * 1024;
|
|
81
|
+
function inflated(data) {
|
|
82
|
+
try { return zlib.inflateSync(data, { maxOutputLength: INFLATE_CAP }).toString('latin1'); } catch (_e) { return null; }
|
|
83
|
+
}
|
|
84
|
+
function pngTextBearing(buf) {
|
|
85
|
+
const parts = [];
|
|
86
|
+
const rest = (from) => { parts.push(buf.slice(from).toString('latin1')); return parts.join('\n'); };
|
|
87
|
+
let off = 8;
|
|
88
|
+
while (off + 8 <= buf.length) {
|
|
89
|
+
const len = buf.readUInt32BE(off);
|
|
90
|
+
const type = buf.slice(off + 4, off + 8).toString('latin1');
|
|
91
|
+
const start = off + 8;
|
|
92
|
+
const end = start + len;
|
|
93
|
+
if (end + 4 > buf.length) return rest(off); // a truncated chunk: it and all after it, whole
|
|
94
|
+
const data = buf.slice(start, end);
|
|
95
|
+
if (type !== 'IDAT') parts.push(type, data.toString('latin1'));
|
|
96
|
+
if (type === 'zTXt') parts.push(zTXtText(data));
|
|
97
|
+
if (type === 'iTXt') parts.push(iTXtText(data));
|
|
98
|
+
off = end + 4;
|
|
99
|
+
if (type === 'IEND') return rest(off); // bytes after IEND, whole
|
|
100
|
+
}
|
|
101
|
+
return rest(off); // fewer than a chunk header's bytes, no IEND
|
|
102
|
+
}
|
|
103
|
+
// keyword \0 method data
|
|
104
|
+
function zTXtText(data) {
|
|
105
|
+
const z = data.indexOf(0);
|
|
106
|
+
return (z >= 0 && inflated(data.slice(z + 2))) || '';
|
|
107
|
+
}
|
|
108
|
+
// keyword \0 flag method language \0 translated keyword \0 text; the text is deflated when flag is 1
|
|
109
|
+
function iTXtText(data) {
|
|
110
|
+
const z = data.indexOf(0);
|
|
111
|
+
if (z < 0 || data[z + 1] !== 1) return '';
|
|
112
|
+
const lang = data.indexOf(0, z + 3);
|
|
113
|
+
const tr = lang >= 0 ? data.indexOf(0, lang + 1) : -1;
|
|
114
|
+
return (tr >= 0 && inflated(data.slice(tr + 1))) || '';
|
|
115
|
+
}
|
|
116
|
+
|
|
57
117
|
function scanContent(rel, content) {
|
|
118
|
+
// A reader may hand this bytes rather than text. A PNG is then read as its text-bearing regions;
|
|
119
|
+
// anything else is decoded and scanned whole.
|
|
120
|
+
if (Buffer.isBuffer(content)) {
|
|
121
|
+
return scanContent(rel, content.slice(0, 8).equals(PNG_SIG) ? pngTextBearing(content) : content.toString('utf8'));
|
|
122
|
+
}
|
|
58
123
|
const hits = [];
|
|
59
124
|
for (const p of PATTERNS) {
|
|
60
125
|
const m = content.match(p.re);
|
|
@@ -104,10 +169,10 @@ function scanFiles(files, read, readLink) {
|
|
|
104
169
|
}
|
|
105
170
|
let c;
|
|
106
171
|
try { c = read(rel); } catch (_e) { continue; }
|
|
107
|
-
if (typeof c !== 'string') continue;
|
|
172
|
+
if (typeof c !== 'string' && !Buffer.isBuffer(c)) continue;
|
|
108
173
|
hits.push(...scanContent(rel, c));
|
|
109
174
|
}
|
|
110
175
|
return hits;
|
|
111
176
|
}
|
|
112
177
|
|
|
113
|
-
module.exports = { PATTERNS, EMAIL_ALLOW, HOSTIP, scanPath, scanContent, scanFiles };
|
|
178
|
+
module.exports = { PATTERNS, EMAIL_ALLOW, HOSTIP, PNG_SIG, pngTextBearing, scanPath, scanContent, scanFiles };
|
package/lib/receipt.js
CHANGED
|
@@ -25,7 +25,11 @@ const SCHEMA_FILES = {
|
|
|
25
25
|
// the validator (it requires the receipt to say what answered it), so the
|
|
26
26
|
// number has to keep resolving to the schema it meant.
|
|
27
27
|
'0.5': 'receipt.v0.5.schema.json',
|
|
28
|
-
|
|
28
|
+
// v0.6 moved the same way when v0.7 took the current pointer (spec 035): v0.7 adds a
|
|
29
|
+
// verdict token a reader derives (UNDERPOWERED), and a v0.6 receipt restamped 0.7
|
|
30
|
+
// is refused by the const, so the number keeps resolving to the schema it meant.
|
|
31
|
+
'0.6': 'receipt.v0.6.schema.json',
|
|
32
|
+
'0.7': 'receipt.schema.json',
|
|
29
33
|
};
|
|
30
34
|
|
|
31
35
|
// THE BAND RULE, stated on every receipt (results.aggregates.band_rule, spec
|