driftproof 0.10.2 → 0.11.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -1,12 +1,13 @@
1
1
  <!-- SPDX-License-Identifier: Apache-2.0 -->
2
2
  # Driftproof
3
3
 
4
- **Your skill passed. On which model? On what date?**
4
+ **Is the gap your skill makes real, or noise? Did it hold on the last model release?**
5
5
 
6
- A SKILL.md teaches an AI coding agent how you like things done. When a new model
7
- ships, the same file can stop helping, or start hurting. Driftproof re-runs the
8
- skill's tests on the new model and hands you a dated, hash-verified receipt
9
- saying whether it still helps.
6
+ A SKILL.md teaches an AI coding agent how you like things done. A skill's own
7
+ tests passing is one answer. Driftproof asks two more: whether the scores with
8
+ the skill and without it separate beyond their spread, and whether that held
9
+ when the model changed. Each answer is a dated, hash-verified receipt, and when
10
+ there were too few draws to tell, the receipt says so.
10
11
 
11
12
  [![driftproof](https://img.shields.io/endpoint?url=https://driftproofhq.com/badges/commit-message-conventions.json)](https://driftproofhq.com)
12
13
  &nbsp;— live badge for the bundled `commit-message-conventions` example, generated from its own receipt.
@@ -295,7 +296,7 @@ A receipt is the unit of evidence — one JSON document conforming to
295
296
 
296
297
  ```jsonc
297
298
  {
298
- "schema_version": "0.6",
299
+ "schema_version": "0.7",
299
300
  "skill": { "name": "commit-message-conventions", "version": "0.2.0",
300
301
  "content_hash": "…sha256 over SKILL.md + bundled files…" },
301
302
  "suite": { "format": "agentskills.io/evals", "suite_hash": "…", "case_count": 10 },
@@ -304,7 +305,7 @@ A receipt is the unit of evidence — one JSON document conforming to
304
305
  "model_release_date": "2025-10-01",
305
306
  "provider": "anthropic",
306
307
  "surface": "claude-cli",
307
- "runner_version": "0.10.2",
308
+ "runner_version": "0.11.1",
308
309
  "date_utc": "2026-07-27T…Z",
309
310
  "registry": "registered",
310
311
  "transcripts": "hashes-only",
@@ -383,7 +384,7 @@ jobs:
383
384
  runs-on: ubuntu-latest
384
385
  steps:
385
386
  - uses: actions/checkout@v4
386
- - uses: driftproofhq/driftproof@v0.10.2
387
+ - uses: driftproofhq/driftproof@v0.11.1
387
388
  with:
388
389
  skill-dir: skills/my-skill
389
390
  models: claude-haiku-4-5
package/bin/driftproof CHANGED
@@ -13,7 +13,7 @@ const { buildDriftReport, revisionPairProblem } = require('../lib/diff');
13
13
  const { surfaceForModel, isSubscriptionSurface, resolveModel, evalUser } = require('../lib/provider');
14
14
  const { estimateRunCostUSD, BudgetTracker } = require('../lib/cost');
15
15
  const { registryStatus, assertRegistered, REGISTRY_PATH } = require('../lib/models');
16
- const { verdictFromReceipt, badgeEndpoint, githubOutputLines } = require('../lib/verdict');
16
+ const { verdictFromReceipt, badgeEndpoint, githubOutputLines, drawsLine, UNDERPOWERED_LINE } = require('../lib/verdict');
17
17
  const decision = require('../lib/decision');
18
18
  const { scaffoldInit } = require('../lib/init');
19
19
 
@@ -75,7 +75,7 @@ function loadRc(skillDir) {
75
75
 
76
76
  // Flags that never take a value, so `--trusted-skill <skill-dir>` keeps the dir
77
77
  // positional instead of swallowing it as the flag's value.
78
- const BOOLEAN_FLAGS = new Set(['trusted-skill']);
78
+ const BOOLEAN_FLAGS = new Set(['trusted-skill', 'svg']);
79
79
 
80
80
  // ── THE INPUT CONTRACT (spec 026 AC-9, F5) ────────────────────────────────────
81
81
  // A COPY of action/lib.sh's RE_* rules, in the shape spec 028's fixture holds
@@ -177,7 +177,7 @@ USAGE
177
177
  [--keep-transcripts] [--out DIR] [--trusted-skill]
178
178
  ${PROJECT_NAME} diff <receiptA.json> <receiptB.json> [--out FILE] [--mode release|revision]
179
179
  ${PROJECT_NAME} validate <receipt.json>
180
- ${PROJECT_NAME} badge <receipt.json|receipts-dir> [--models a,b] [--out FILE] [--github-output]
180
+ ${PROJECT_NAME} badge <receipt.json|receipts-dir> [--models a,b] [--out FILE] [--github-output] [--svg [--href URL]]
181
181
  ${PROJECT_NAME} decide <receipts-dir> --models a,b [--github-output] [--badge FILE]
182
182
  [--summary FILE] [--enforce] [--fail-on-regression true|false]
183
183
  ${PROJECT_NAME} import <results.json> --from agent-skills-eval|skillgrade [--out DIR]
@@ -515,17 +515,35 @@ function cmdBadge(positional, flags) {
515
515
  if (fs.existsSync(p) && fs.statSync(p).isDirectory()) return cmdBadgeSet(p, flags);
516
516
  const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
517
517
  refuseUnverified(receipt, p, 'badge');
518
+ if (flags.svg) {
519
+ // SPEC 036. The drawn badge: state, model, date, lift and uncertainty, with the
520
+ // machine token in its data attributes. After the same hash check as the JSON.
521
+ const svg = require('../lib/badge-svg').badgeSvg(receipt, { href: typeof flags.href === 'string' ? flags.href : null });
522
+ if (flags.out) {
523
+ const out = path.resolve(flags.out);
524
+ fs.mkdirSync(path.dirname(out), { recursive: true });
525
+ fs.writeFileSync(out, svg);
526
+ console.log(`badge written to ${flags.out} (${verdictFromReceipt(receipt).verdict}, svg)`);
527
+ } else process.stdout.write(svg);
528
+ return;
529
+ }
518
530
  if (flags['github-output']) { console.log(githubOutputLines(receipt)); return; }
519
531
  const badge = badgeEndpoint(receipt);
520
532
  const json = JSON.stringify(badge, null, 2);
533
+ const v = verdictFromReceipt(receipt);
534
+ // SPEC 035 AC-3. An UNDERPOWERED receipt says so in words and says what it would
535
+ // have needed. Beside JSON on stdout the two lines go to stderr, so a caller that
536
+ // redirects the badge still gets a file that parses.
537
+ const power = v.verdict === 'UNDERPOWERED' ? [`${UNDERPOWERED_LINE}.`, drawsLine(v.drawsNeeded)] : [];
521
538
  if (flags.out) {
522
539
  const out = path.resolve(flags.out);
523
540
  fs.mkdirSync(path.dirname(out), { recursive: true });
524
541
  fs.writeFileSync(out, json + '\n');
525
- const v = verdictFromReceipt(receipt);
526
542
  console.log(`badge written to ${flags.out} (${v.verdict}: ${badge.message}, ${badge.color})`);
543
+ for (const line of power) console.log(line);
527
544
  } else {
528
545
  console.log(json);
546
+ for (const line of power) console.error(line);
529
547
  }
530
548
  }
531
549
 
package/config.js CHANGED
@@ -9,7 +9,7 @@ const PROJECT_NAME = 'driftproof';
9
9
  // Bumped whenever the runner's behaviour or receipt-generation semantics change
10
10
  // in a way that could affect results. Recorded into every receipt as
11
11
  // run.runner_version so a receipt is reproducible against a known engine.
12
- const RUNNER_VERSION = '0.10.2';
12
+ const RUNNER_VERSION = '0.11.1';
13
13
 
14
14
  // The eval format we CONSUME (we deliberately do not invent our own).
15
15
  const SUITE_FORMAT = 'agentskills.io/evals';
@@ -38,7 +38,7 @@ const SUITE_FORMAT = 'agentskills.io/evals';
38
38
  // results.aggregates.band_rule, and bands that are null where the formula
39
39
  // cannot form. NOT additive for the validator: a v0.5 receipt restamped 0.6
40
40
  // is refused. v0.5 is frozen as receipt.v0.5.schema.json.
41
- const RECEIPT_SCHEMA_VERSION = '0.6';
41
+ const RECEIPT_SCHEMA_VERSION = '0.7';
42
42
 
43
43
  // Input bounds on the skill loader and the post-checks (spec 026 AC-13, audit
44
44
  // A3/A7). lib/skill.js loaded every bundled file with no count, depth, byte,
@@ -104,6 +104,13 @@ const DEFAULT_JUDGE_SAMPLES = 5;
104
104
  // spec/RECEIPT.md § "Drift verdict rule" and the report methodology.
105
105
  const EFFECT_FLOOR = 0.05;
106
106
 
107
+ // How many standard errors of the difference between two means a case's resolution
108
+ // adds to its two spreads (spec 035 R-2). A case the band rule does not separate is
109
+ // UNDERPOWERED when its spreads plus this many standard errors reach the floor: at
110
+ // those spreads and draws, a true shift of the floor's size could have gone unseparated.
111
+ // Named, so a later decision moves it by one line and one amendment.
112
+ const POWER_Z = 2;
113
+
107
114
  // ── generation sampling (receipt spec v0.5) ─────────────────────────────────
108
115
  //
109
116
  // Report #006 measured across-draw spread at sd 0.186 and 0.183 while the
@@ -157,7 +164,7 @@ const REPORT_005_JUDGE_MODEL = 'claude-haiku-4-5';
157
164
  module.exports = {
158
165
  PROJECT_NAME, RUNNER_VERSION, SUITE_FORMAT, RECEIPT_SCHEMA_VERSION, DEFAULT_JUDGE_SAMPLES,
159
166
  SKILL_MAX_FILES, SKILL_MAX_BYTES, SKILL_MAX_DEPTH, SUITE_MAX_CASES, CASE_MAX_CHARS, CHECK_MAX_PATTERN,
160
- EFFECT_FLOOR, DEV_MAX_USD, DEV_MAX_CALLS, REPORT_MAX_USD, TRIGGER_MAX_USD,
167
+ EFFECT_FLOOR, POWER_Z, DEV_MAX_USD, DEV_MAX_CALLS, REPORT_MAX_USD, TRIGGER_MAX_USD,
161
168
  GENERATION_SAMPLES_MIN, GENERATION_SAMPLES_MAX, GENERATION_SD_THRESHOLD, GENERATION_STABILITY_EPS,
162
169
  REPORT_002_CLAUDE_MODEL, REPORT_002_GPT_MODEL, REPORT_002_JUDGE_MODEL,
163
170
  REPORT_003_NEW_MODEL, REPORT_003_OLD_MODEL, REPORT_003_JUDGE_MODEL,
@@ -0,0 +1,130 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // The badge, drawn (spec 036).
5
+ //
6
+ // The shields endpoint JSON (lib/verdict.js badgeEndpoint) carries a label, a message
7
+ // and a colour, which is all a shields badge can hold. This draws the badge itself, so
8
+ // it can show what a reader needs at a glance: the state, the model, the date, the lift
9
+ // and its uncertainty. Four segments, left to right:
10
+ //
11
+ // driftproof | <state label> | <model> · <date> | lift <+0.000> <± 0.000 | no ±>
12
+ //
13
+ // The fourth segment carries the lift ONLY for a state that measured one. UNDERPOWERED and
14
+ // NOT_MEASURED carry the draws taken against the draws needed instead (A-036-2):
15
+ //
16
+ // driftproof | not enough draws | <model> · <date> | 3 draws of 87 needed
17
+ //
18
+ // THE WORDS ARE HUMAN; THE TOKENS ARE DATA. What a reader sees is a label (passing, not
19
+ // enough draws); the machine token the Action and the differ publish sits on the root
20
+ // element as data-verdict, with the receipt hash beside it, and never in the text.
21
+ //
22
+ // UNDERPOWERED IS DRAWN AS THE HONEST STATE, not as a failure: a calm indigo no other
23
+ // state uses, an open hatch over it, a dashed edge around it, and the words "not enough
24
+ // draws". Every state's text is white on a fill it clears 4.5:1 against, hatch included.
25
+ //
26
+ // Text widths are fixed with textLength, so a text's rendered box is the width this file
27
+ // allots it whatever font the viewer has, and it cannot spill out of its segment.
28
+
29
+ const { receiptVerdict, shortModel, receiptDrawsTaken } = require('./verdict');
30
+
31
+ const INK = '#ffffff';
32
+ const BRAND = '#24292f';
33
+ const DETAIL = '#4b5563';
34
+ const EFFECT = '#374151';
35
+ const STATE = {
36
+ PASSED: { fill: '#1a7f37' },
37
+ REGRESSED: { fill: '#b42318' },
38
+ UNDERPOWERED: { fill: '#3538cd', hatch: '#444ce7', edge: '#a4bcfd' },
39
+ NO_EFFECT: { fill: '#57606a' },
40
+ NOT_MEASURED: { fill: '#656d76' },
41
+ };
42
+ const LABELS = {
43
+ PASSED: 'passing',
44
+ REGRESSED: 'regressed',
45
+ UNDERPOWERED: 'not enough draws',
46
+ NO_EFFECT: 'no separation detected',
47
+ NOT_MEASURED: 'not measured',
48
+ };
49
+
50
+ const H = 20;
51
+ const PAD = 7;
52
+ const CHAR = 6.6;
53
+ const esc = (s) => String(s).replace(/&/g, '&amp;').replace(/</g, '&lt;').replace(/>/g, '&gt;').replace(/"/g, '&quot;');
54
+ const textW = (s) => Math.round(String(s).length * CHAR * 10) / 10;
55
+
56
+ // A LIFT FIGURE IS A MEASUREMENT, so only a state that measured one may print it
57
+ // (spec 036 A-036-2). UNDERPOWERED says the receipt's draws cannot resolve a shift of the
58
+ // effect floor, and NOT_MEASURED says the receipt carries no verdict at all. A lift beside
59
+ // either is the badge answering, in its most legible segment, a question the state has just
60
+ // said it cannot answer -- and a reader who takes one figure from a badge takes that one.
61
+ // Those two states print what the reader can act on instead: the draws taken against the
62
+ // draws that would have been needed.
63
+ function effectText(receipt, v) {
64
+ const c = receipt.comparison || {};
65
+ if (v.verdict === 'UNDERPOWERED' || v.verdict === 'NOT_MEASURED') return drawsText(receipt, v);
66
+ const lift = typeof c.delta === 'number' ? `lift ${c.delta >= 0 ? '+' : '-'}${Math.abs(c.delta).toFixed(3)}` : 'no lift';
67
+ const unc = typeof c.delta_uncertainty === 'number' ? `± ${c.delta_uncertainty.toFixed(3)}`
68
+ : c.delta_uncertainty_unavailable === 'single_case' ? 'no ± (1 case)' : 'no ±';
69
+ return `${lift} ${unc}`;
70
+ }
71
+
72
+ // Draws taken against draws needed, in the badge's words. The needed count is
73
+ // receiptDrawsNeeded's, the same figure drawsLine prints in a sentence, so the badge and
74
+ // the page cannot say two different numbers. Where no count exists the badge says which
75
+ // of the two reasons it is, rather than printing a bare figure or nothing at all.
76
+ //
77
+ // BOTH NUMBERS ARE ONE CASE'S (A-036-7). Where a case sets the draws needed, the draws taken
78
+ // are that case's own, which receiptDrawsNeeded carries as `taken`. This read the smallest
79
+ // over every readable case, so a receipt whose binding case drew 4 and whose other case drew
80
+ // 2 badged "2 draws of 5 needed": a 2 from a case the 5 does not belong to. Where no case sets
81
+ // the draws needed (NOT_MEASURED carries no verdict to compute one from) there is no binding
82
+ // case, and the smallest over every readable case is the figure.
83
+ function drawsText(receipt, v) {
84
+ const d = v.drawsNeeded;
85
+ const taken = d ? d.taken : receiptDrawsTaken(receipt);
86
+ if (taken === null || taken === undefined) return 'no draws recorded';
87
+ const draws = `${taken} ${taken === 1 ? 'draw' : 'draws'}`;
88
+ if (!d) return `${draws}, needed not computed`;
89
+ if (typeof d.value === 'number') return `${draws} of ${d.value} needed`;
90
+ return `${draws}, no count at these spreads`;
91
+ }
92
+
93
+ function badgeSvg(receipt, { href = null } = {}) {
94
+ const v = receiptVerdict(receipt);
95
+ const state = STATE[v.verdict];
96
+ const label = LABELS[v.verdict];
97
+ const model = shortModel((receipt.run || {}).model_id);
98
+ const date = String((receipt.run || {}).date_utc || '').slice(0, 10);
99
+ const segs = [
100
+ { seg: 'brand', text: 'driftproof', fill: BRAND },
101
+ { seg: 'state', text: label, fill: state.fill },
102
+ { seg: 'detail', text: `${model} · ${date}`, fill: DETAIL },
103
+ { seg: 'effect', text: effectText(receipt, v), fill: EFFECT },
104
+ ];
105
+ let x = 0;
106
+ for (const s of segs) { s.w = Math.round(textW(s.text) + 2 * PAD); s.x = x; x += s.w; }
107
+ const width = x;
108
+ const title = `driftproof: ${label} on ${model}, ${date}, ${effectText(receipt, v)}`;
109
+ const L = [];
110
+ L.push(`<svg xmlns="http://www.w3.org/2000/svg" width="${width}" height="${H}" viewBox="0 0 ${width} ${H}" role="img" aria-label="${esc(title)}" data-verdict="${v.verdict}" data-receipt-hash="${esc(receipt.receipt_hash || '')}">`);
111
+ L.push(`<title>${esc(title)}</title>`);
112
+ if (state.hatch) {
113
+ L.push(`<defs><pattern id="hatch" width="6" height="6" patternUnits="userSpaceOnUse" patternTransform="rotate(45)"><line x1="0" y1="0" x2="0" y2="6" stroke="${state.hatch}" stroke-width="3"/></pattern></defs>`);
114
+ }
115
+ if (href) L.push(`<a href="${esc(href)}">`);
116
+ for (const s of segs) {
117
+ const isState = s.seg === 'state';
118
+ const edge = isState && state.edge ? ` stroke="${state.edge}" stroke-width="1" stroke-dasharray="3 2"` : '';
119
+ L.push(`<rect data-seg="${s.seg}" x="${s.x}" y="0" width="${s.w}" height="${H}" fill="${s.fill}"${edge}/>`);
120
+ if (isState && state.hatch) L.push(`<rect data-seg="state-hatch" x="${s.x}" y="0" width="${s.w}" height="${H}" fill="url(#hatch)" opacity="0.35"/>`);
121
+ }
122
+ for (const s of segs) {
123
+ L.push(`<text data-seg="${s.seg}" x="${s.x + PAD}" y="14" fill="${INK}" font-family="Verdana,DejaVu Sans,sans-serif" font-size="11" textLength="${textW(s.text)}" lengthAdjust="spacingAndGlyphs">${esc(s.text)}</text>`);
124
+ }
125
+ if (href) L.push('</a>');
126
+ L.push('</svg>');
127
+ return L.join('\n') + '\n';
128
+ }
129
+
130
+ module.exports = { badgeSvg, LABELS, STATE };
package/lib/decision.js CHANGED
@@ -20,7 +20,7 @@
20
20
  const fs = require('fs');
21
21
  const path = require('path');
22
22
  const { EFFECT_FLOOR } = require('../config');
23
- const { shortModel, githubOutputEntry } = require('./verdict');
23
+ const { shortModel, githubOutputEntry, receiptVerdict, drawsLine, UNDERPOWERED_LINE } = require('./verdict');
24
24
 
25
25
  // The six decision states (spec 030 AC-4).
26
26
  //
@@ -29,7 +29,12 @@ const { shortModel, githubOutputEntry } = require('./verdict');
29
29
  // that says so. `no detected effect` is false about such a model - the effect
30
30
  // was detected and it cleared the floor - and widening it to mean "not a
31
31
  // regression" would put a measured lift and a measured nothing in one cell.
32
- const STATES = ['regression', 'refused', 'inconclusive', 'not measured', 'no detected effect', 'helped'];
32
+ //
33
+ // `underpowered` is the seventh (spec 035): the receipt measured, and its own spreads
34
+ // and draws could not resolve a shift of the effect floor. It sits after `not
35
+ // measured`, which measured nothing, and before `no detected effect`, which measured
36
+ // with a resolution below the floor.
37
+ const STATES = ['regression', 'refused', 'inconclusive', 'not measured', 'underpowered', 'no detected effect', 'helped'];
33
38
 
34
39
  // WORST FIRST. This is the single definition of the ordering; AC-1's
35
40
  // enforcement and AC-3's badge both read it, so they cannot disagree about
@@ -38,7 +43,7 @@ const STATE_ORDER = STATES.slice();
38
43
 
39
44
  // States that must never render as success on any surface the action writes -
40
45
  // the badge, the summary row, or the check title (AC-4).
41
- const NEVER_SUCCESS = ['inconclusive', 'not measured'];
46
+ const NEVER_SUCCESS = ['inconclusive', 'not measured', 'underpowered'];
42
47
 
43
48
  // States that fail the job. `refused` fails CLOSED (AC-2): a model that
44
49
  // produced no receipt is not a model that passed. `regression` fails subject to
@@ -86,11 +91,15 @@ function decisionState(receipt) {
86
91
  if (level !== 'TESTED' || kind !== 'model') return 'not measured';
87
92
  const cmp = receipt.comparison || {};
88
93
  if ((receipt.run && receipt.run.status === 'incomplete') || typeof cmp.delta !== 'number') return 'inconclusive';
89
- if (cmp.delta <= -EFFECT_FLOOR) return 'regression';
90
- if (cmp.delta >= EFFECT_FLOOR) return 'helped';
91
- return 'no detected effect';
94
+ // Spec 035: the measured states are the receipt verdict's, read per case by the
95
+ // band rule (lib/verdict.js receiptVerdict), not the aggregate delta against the floor.
96
+ // receiptVerdict refuses on the same four conditions the two clauses above split
97
+ // between `not measured` and `inconclusive`, so it cannot answer NOT_MEASURED here.
98
+ return MEASURED_STATE[receiptVerdict(receipt).verdict];
92
99
  }
93
100
 
101
+ const MEASURED_STATE = { REGRESSED: 'regression', PASSED: 'helped', UNDERPOWERED: 'underpowered', NO_EFFECT: 'no detected effect' };
102
+
94
103
  // The worst state in a set, by STATE_ORDER. An empty set has no decision, and
95
104
  // says so with null rather than defaulting to something benign.
96
105
  function worstState(states) {
@@ -182,6 +191,8 @@ function decideSet(dir, requestedCsv, { failOnRegression = true } = {}) {
182
191
  delta: receipt && receipt.comparison && typeof receipt.comparison.delta === 'number'
183
192
  ? receipt.comparison.delta : null,
184
193
  file: hit ? path.basename(hit.file) : (unreadable ? path.basename(unreadable.file) : null),
194
+ // Spec 035 AC-3: what an underpowered row would have needed, from the same reading.
195
+ drawsNeeded: state === 'underpowered' ? receiptVerdict(receipt).drawsNeeded : null,
185
196
  // The distinction is carried on the ROW, not left in a sentence, because
186
197
  // the enforcement message has to say which of the two happened (AC-2's
187
198
  // mutation class, absence-vs-unreadable).
@@ -246,6 +257,7 @@ const RENDER = {
246
257
  refused: { word: 'refused', color: 'red', marker: '\u274c' },
247
258
  inconclusive: { word: 'inconclusive', color: 'yellow', marker: '\u26a0\ufe0f' },
248
259
  'not measured': { word: 'not measured', color: 'lightgrey', marker: '\u26a0\ufe0f' },
260
+ underpowered: { word: 'not enough draws', color: 'blue', marker: '\u26a0\ufe0f' },
249
261
  'no detected effect': { word: 'no effect', color: 'lightgrey', marker: '\u2014' },
250
262
  helped: { word: 'passing', color: 'brightgreen', marker: '\u2705' },
251
263
  };
@@ -283,7 +295,7 @@ function badgeEndpointForSet(d, { label = 'driftproof' } = {}) {
283
295
  // `worst`, `regressed` and `missing` are new and carry what it could not say.
284
296
  const VERDICT_WORD = {
285
297
  regression: 'REGRESSED', refused: 'REFUSED', inconclusive: 'INCONCLUSIVE',
286
- 'not measured': 'NOT_MEASURED', 'no detected effect': 'NO_EFFECT', helped: 'PASSED',
298
+ 'not measured': 'NOT_MEASURED', underpowered: 'UNDERPOWERED', 'no detected effect': 'NO_EFFECT', helped: 'PASSED',
287
299
  };
288
300
 
289
301
  function githubOutputLines(d) {
@@ -299,6 +311,10 @@ function githubOutputLines(d) {
299
311
  ['regressed_models', d.regressed.join(',')],
300
312
  ['missing_models', d.missing.join(',')],
301
313
  ['receipt_count', d.receiptCount],
314
+ // Spec 035 AC-3: the draws the worst row would have needed, when that row is
315
+ // underpowered; `none` when no draw count reaches the floor at its spreads, and
316
+ // empty otherwise.
317
+ ['draws_needed', worstRow && worstRow.drawsNeeded ? (worstRow.drawsNeeded.value === null ? 'none' : worstRow.drawsNeeded.value) : ''],
302
318
  ].map(([k, v]) => githubOutputEntry(k, v)).join('\n');
303
319
  }
304
320
 
@@ -324,7 +340,8 @@ function summaryMarkdown(d) {
324
340
  const r = RENDER[row.state];
325
341
  const delta = row.delta === null ? 'n/a' : (row.delta >= 0 ? '+' : '') + row.delta.toFixed(3);
326
342
  const note = row.reason ? ` <br><sub>${cell(row.reason)}</sub>` : '';
327
- L.push(`| \`${cell(row.model)}\` | ${r.marker} ${row.state} | ${delta} | ${cell(row.file) || '\u2014'}${note} |`);
343
+ const power = row.state === 'underpowered' ? ` <br><sub>${cell(UNDERPOWERED_LINE)}. ${cell(drawsLine(row.drawsNeeded))}</sub>` : '';
344
+ L.push(`| \`${cell(row.model)}\` | ${r.marker} ${row.state}${power} | ${delta} | ${cell(row.file) || '\u2014'}${note} |`);
328
345
  }
329
346
  L.push('');
330
347
  L.push(`Decided over ${d.receiptCount} receipt(s) for ${d.requestedCount} requested model(s). `
package/lib/diff.js CHANGED
@@ -5,6 +5,7 @@ const { bandVerdict, round, WITHIN_NOISE } = require('./stats');
5
5
  const { EFFECT_FLOOR } = require('../config');
6
6
  const { revisionHeadline } = require('./revision');
7
7
  const { baselineReproduces, REFUSAL_REASONS, bandOf } = require('./reuse');
8
+ const { caseRule, armOf, drawsLine, UNDERPOWERED_LINE } = require('./verdict');
8
9
 
9
10
  // Practical-significance gate applied ON TOP of band separation. bandVerdict()
10
11
  // stays a pure geometry test (kept that way so its unit checks are unambiguous);
@@ -198,6 +199,15 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
198
199
  // untested input does: no delta is asserted, and the reason travels with it.
199
200
  const measured = belowTested.length === 0 && !refusal && incompleteSides.length === 0 && !judgeProblem;
200
201
 
202
+ // THE UNDERPOWERED STATE RIDES BESIDE THE VERDICT VALUE (spec 035 R-6). The value
203
+ // is unchanged, because Reports 006 to 008 and spec 009's gate compare against it
204
+ // and a published record quotes the lines it produced. `power` says, for a case
205
+ // the rule did not separate, whether the two with_skill arms' own spreads and
206
+ // draws could have resolved a shift of the effect floor, by the rule lib/verdict.js
207
+ // applies to a single receipt.
208
+ const rawWith = (r) => new Map(r.results.cases.filter((c) => c.mode === 'with_skill').map((c) => [c.id, c]));
209
+ const aRaw = rawWith(a);
210
+ const bRaw = rawWith(b);
201
211
  const perCase = ids.map((id) => {
202
212
  const before = aB[id] || null;
203
213
  const after = bB[id] || null;
@@ -205,7 +215,13 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
205
215
  const verdict = refusal ? 'refused'
206
216
  : !measured ? 'not measured'
207
217
  : (before && after) ? verdictWithFloor(before, after, delta) : 'n/a';
208
- return { id, before, after, delta, verdict };
218
+ let power = null;
219
+ if (measured && before && after && (verdict === WITHIN_NOISE || verdict === WITHIN_NOISE_FLOOR)) {
220
+ const ra = armOf(aRaw.get(id)); const rb = armOf(bRaw.get(id));
221
+ const rule = ra && rb ? caseRule(ra, rb) : null;
222
+ if (rule && rule.state === 'underpowered') power = { state: 'underpowered', drawsNeeded: rule.drawsNeeded, reason: rule.reason, ...(rule.spread !== undefined ? { spread: rule.spread } : {}) };
223
+ }
224
+ return { id, before, after, delta, verdict, power };
209
225
  });
210
226
  // Sort worst-first: regressions, then by delta.
211
227
  const order = { regression: 0, [WITHIN_NOISE]: 1, [WITHIN_NOISE_FLOOR]: 1, improvement: 2, 'n/a': 3, 'not measured': 3, refused: 3 };
@@ -391,6 +407,18 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
391
407
  // the rule, which is what the headline above says.
392
408
  L.push(`with_skill mean moved ${fmt(headlineDelta)} (${bandStr(aAgg)} → ${bandStr(bAgg)}; band = suite dispersion). Per-case band-overlap verdicts: ${regressions.length} regression(s), ${perCase.filter((r) => r.verdict === 'improvement').length} improvement(s), ${nWithin - nFloor} with no separation detected${nFloor ? `, ${nFloor} below the ${EFFECT_FLOOR} effect floor` : ''}.`);
393
409
  L.push('');
410
+ const underpowered = perCase.filter((r) => r.power);
411
+ if (underpowered.length) {
412
+ // One count, one sentence, one draws line: the worst case's (spec 035 AC-5).
413
+ const spread = underpowered.find((r) => r.power.reason === 'spread');
414
+ const most = underpowered.filter((r) => r.power.reason === 'draws').sort((x, y) => y.power.drawsNeeded - x.power.drawsNeeded)[0];
415
+ const worst = spread ? { value: null, reason: 'spread', case: spread.id, spread: spread.power.spread }
416
+ : most ? { value: most.power.drawsNeeded, reason: 'draws', case: most.id }
417
+ : { value: 2, reason: 'single_draw', case: underpowered[0].id };
418
+ const powerLine = `Underpowered: ${underpowered.length} of the cases with no separation under the rule. ${UNDERPOWERED_LINE}. ${drawsLine(worst)}`;
419
+ L.push(powerLine);
420
+ L.push('');
421
+ }
394
422
  }
395
423
 
396
424
  L.push(`## Per-case with_skill (band overlap → verdict)`);
@@ -399,7 +427,8 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
399
427
  L.push(`|---|---|---|---|---|`);
400
428
  for (const r of perCase) {
401
429
  const flag = r.verdict === 'regression' ? '🔻 regression' : r.verdict === 'improvement' ? '🔼 improvement' : r.verdict === WITHIN_NOISE_FLOOR ? 'below effect floor' : r.verdict === WITHIN_NOISE ? 'no separation detected' : r.verdict === 'not measured' ? 'not measured' : 'n/a';
402
- L.push(`| \`${r.id}\` | ${r.before ? bandStr(r.before) : 'n/a'} | ${r.after ? bandStr(r.after) : 'n/a'} | ${fmt(r.delta)} | ${flag} |`);
430
+ const label = r.power ? `${flag}; ${UNDERPOWERED_LINE}` : flag;
431
+ L.push(`| \`${r.id}\` | ${r.before ? bandStr(r.before) : 'n/a'} | ${r.after ? bandStr(r.after) : 'n/a'} | ${fmt(r.delta)} | ${label} |`);
403
432
  }
404
433
  L.push('');
405
434
 
package/lib/hygiene.js CHANGED
@@ -23,6 +23,8 @@
23
23
  // Patterns are written so a regex literal cannot match its own text: a
24
24
  // metacharacter or a character class follows each fixed prefix.
25
25
 
26
+ const zlib = require('zlib');
27
+
26
28
  const dec = (b64) => JSON.parse(Buffer.from(b64, 'base64').toString('utf8'));
27
29
 
28
30
  // Build-box host + private IP, base64 so the plaintext never lands in a file.
@@ -54,7 +56,70 @@ function scanPath(rel) {
54
56
  return /(^|\/)\.env(\.|$)/.test(rel) ? [{ file: rel, kind: 'env-file' }] : [];
55
57
  }
56
58
 
59
+ // A PNG'S TEXT-BEARING REGIONS (spec 038 A-038-5, the classification spec 027 already made for the
60
+ // repository gate's confidentiality scan, applied here).
61
+ //
62
+ // THE THREAT MODEL IS ACCIDENTAL PLAINTEXT DISCLOSURE by a person or an agent writing the tree
63
+ // (CONSTITUTION § Threat model). A PNG carries text in its tEXt, iTXt and zTXt chunks, and carries
64
+ // pixels in IDAT, which is deflate output: nobody types into it, and a run of bytes that happens to
65
+ // decode as an address there is a coincidence of compression, not a disclosure. Scanned as one
66
+ // string, a 2 MB screenshot is two million chances of that coincidence, and it has fired twice on
67
+ // this repository's own committed captures.
68
+ //
69
+ // So: every chunk but IDAT is scanned, chunk type names included, and IDAT is not. A name RENDERED
70
+ // INTO a screenshot is pixels, which no byte scan sees either way - that is the gap this scan has
71
+ // always had, and it is unchanged. Everything that is not a PNG is scanned whole, as before.
72
+ //
73
+ // FAIL CLOSED WHERE THE FILE STOPS BEING A PNG (spec 038 approval F-3). Bytes after IEND are no chunk
74
+ // at all, and a chunk whose length runs past the end of the file is a truncated file: either way the
75
+ // remainder, from where the walk lost the structure, is scanned whole, as every file was before this
76
+ // classification. And a text chunk is read in the encoding it carries text in: a zTXt chunk's text
77
+ // and a compressed iTXt chunk's text are inflated and scanned as well as their raw bytes. A chunk
78
+ // that does not inflate is scanned raw, which it already was.
79
+ const PNG_SIG = Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a]);
80
+ const INFLATE_CAP = 16 * 1024 * 1024;
81
+ function inflated(data) {
82
+ try { return zlib.inflateSync(data, { maxOutputLength: INFLATE_CAP }).toString('latin1'); } catch (_e) { return null; }
83
+ }
84
+ function pngTextBearing(buf) {
85
+ const parts = [];
86
+ const rest = (from) => { parts.push(buf.slice(from).toString('latin1')); return parts.join('\n'); };
87
+ let off = 8;
88
+ while (off + 8 <= buf.length) {
89
+ const len = buf.readUInt32BE(off);
90
+ const type = buf.slice(off + 4, off + 8).toString('latin1');
91
+ const start = off + 8;
92
+ const end = start + len;
93
+ if (end + 4 > buf.length) return rest(off); // a truncated chunk: it and all after it, whole
94
+ const data = buf.slice(start, end);
95
+ if (type !== 'IDAT') parts.push(type, data.toString('latin1'));
96
+ if (type === 'zTXt') parts.push(zTXtText(data));
97
+ if (type === 'iTXt') parts.push(iTXtText(data));
98
+ off = end + 4;
99
+ if (type === 'IEND') return rest(off); // bytes after IEND, whole
100
+ }
101
+ return rest(off); // fewer than a chunk header's bytes, no IEND
102
+ }
103
+ // keyword \0 method data
104
+ function zTXtText(data) {
105
+ const z = data.indexOf(0);
106
+ return (z >= 0 && inflated(data.slice(z + 2))) || '';
107
+ }
108
+ // keyword \0 flag method language \0 translated keyword \0 text; the text is deflated when flag is 1
109
+ function iTXtText(data) {
110
+ const z = data.indexOf(0);
111
+ if (z < 0 || data[z + 1] !== 1) return '';
112
+ const lang = data.indexOf(0, z + 3);
113
+ const tr = lang >= 0 ? data.indexOf(0, lang + 1) : -1;
114
+ return (tr >= 0 && inflated(data.slice(tr + 1))) || '';
115
+ }
116
+
57
117
  function scanContent(rel, content) {
118
+ // A reader may hand this bytes rather than text. A PNG is then read as its text-bearing regions;
119
+ // anything else is decoded and scanned whole.
120
+ if (Buffer.isBuffer(content)) {
121
+ return scanContent(rel, content.slice(0, 8).equals(PNG_SIG) ? pngTextBearing(content) : content.toString('utf8'));
122
+ }
58
123
  const hits = [];
59
124
  for (const p of PATTERNS) {
60
125
  const m = content.match(p.re);
@@ -104,10 +169,10 @@ function scanFiles(files, read, readLink) {
104
169
  }
105
170
  let c;
106
171
  try { c = read(rel); } catch (_e) { continue; }
107
- if (typeof c !== 'string') continue;
172
+ if (typeof c !== 'string' && !Buffer.isBuffer(c)) continue;
108
173
  hits.push(...scanContent(rel, c));
109
174
  }
110
175
  return hits;
111
176
  }
112
177
 
113
- module.exports = { PATTERNS, EMAIL_ALLOW, HOSTIP, scanPath, scanContent, scanFiles };
178
+ module.exports = { PATTERNS, EMAIL_ALLOW, HOSTIP, PNG_SIG, pngTextBearing, scanPath, scanContent, scanFiles };
package/lib/receipt.js CHANGED
@@ -25,7 +25,11 @@ const SCHEMA_FILES = {
25
25
  // the validator (it requires the receipt to say what answered it), so the
26
26
  // number has to keep resolving to the schema it meant.
27
27
  '0.5': 'receipt.v0.5.schema.json',
28
- '0.6': 'receipt.schema.json',
28
+ // v0.6 moved the same way when v0.7 took the current pointer (spec 035): v0.7 adds a
29
+ // verdict token a reader derives (UNDERPOWERED), and a v0.6 receipt restamped 0.7
30
+ // is refused by the const, so the number keeps resolving to the schema it meant.
31
+ '0.6': 'receipt.v0.6.schema.json',
32
+ '0.7': 'receipt.schema.json',
29
33
  };
30
34
 
31
35
  // THE BAND RULE, stated on every receipt (results.aggregates.band_rule, spec