driftproof 0.5.0 → 0.7.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/diff.js CHANGED
@@ -3,6 +3,8 @@
3
3
 
4
4
  const { bandVerdict, round } = require('./stats');
5
5
  const { EFFECT_FLOOR } = require('../config');
6
+ const { revisionHeadline } = require('./revision');
7
+ const { baselineReproduces, REFUSAL_REASONS, bandOf } = require('./reuse');
6
8
 
7
9
  // Practical-significance gate applied ON TOP of band separation. bandVerdict()
8
10
  // stays a pure geometry test (kept that way so its unit checks are unambiguous);
@@ -31,11 +33,23 @@ function isWithinNoise(v) { return v === 'within noise' || v === WITHIN_NOISE_FL
31
33
 
32
34
  // Map case-id → { mean, stddev } for a receipt's with_skill cases. Falls back to
33
35
  // score/0 for v0.1 receipts that have no per-case band.
36
+ // NO RENDERING SWITCHES. An earlier revision carried two const-true flags so each
37
+ // half of F-015-B's fix could be removed in a probe — and the dead branch of one
38
+ // of them contained, verbatim, the false headline AC-4 exists to forbid. Nothing
39
+ // could fire it, and it was still a defect-restoring branch living in the tree
40
+ // that gets published. The mutation probes patch this source in a disposable copy.
41
+
34
42
  function withSkillBands(receipt) {
35
43
  const out = {};
36
44
  for (const c of receipt.results.cases) {
37
45
  if (c.mode !== 'with_skill') continue;
38
- out[c.id] = { mean: c.mean != null ? c.mean : c.score, stddev: c.stddev || 0 };
46
+ // ONE BAND DEFINITION (spec 016 AC-1). This built its own from the v0.4-shaped
47
+ // fields, which worked — and that is the point: three copies existed, two
48
+ // reading the legacy shape and one reading only v0.5, and the one that read
49
+ // only v0.5 was the one a cross-version control depended on (F-015-C). Routing
50
+ // every comparison path through `bandOf` means a future shape is added once.
51
+ const b = bandOf(c);
52
+ if (b) out[c.id] = { mean: b.mean, stddev: b.sd, source: b.source };
39
53
  }
40
54
  return out;
41
55
  }
@@ -50,7 +64,26 @@ function aggWithBand(receipt) {
50
64
 
51
65
  function fmt(n) { return n == null ? 'n/a' : (n >= 0 ? '+' : '') + n.toFixed(3); }
52
66
  function pct(n) { return n == null ? 'n/a' : n.toFixed(3); }
53
- function bandStr(x) { return `${x.mean.toFixed(3)} ± ${x.stddev.toFixed(3)}`; }
67
+ // THE BAND SAYS WHICH BAND IT IS (spec 017 AC-7).
68
+ //
69
+ // `bandOf` computes `source: 'legacy' | 'generation'` and carries it onto every
70
+ // band; nothing rendered it, so a reader comparing an archived receipt with a
71
+ // v0.5 one was comparing a JUDGE-SAMPLE spread against an ACROSS-DRAW spread
72
+ // with nothing on the page saying so. They are different statistics over
73
+ // different things, and the comparison is still the only one v0.4 admits — which
74
+ // is exactly why the page has to name them rather than leave them to look alike.
75
+ //
76
+ // THE MARKER IS THE RECEIPT'S OWN WORD — `legacy` or `generation`, exactly as
77
+ // `bandOf` records it — rather than a prettier synonym. A reader who greps the
78
+ // page for what a receipt says should find the same token; a rendering that
79
+ // renames the thing it is disclosing has disclosed a different thing. Omitted
80
+ // when a band carries no source (an aggregate band is computed from case means,
81
+ // not from one case's draws).
82
+
83
+ function bandStr(x) {
84
+ const label = x && x.source;
85
+ return `${x.mean.toFixed(3)} ± ${x.stddev.toFixed(3)}${label ? ` (${label})` : ''}`;
86
+ }
54
87
  function short(h) { return h ? String(h).slice(0, 12) : 'n/a'; }
55
88
 
56
89
  // The headline is a SUMMARY of the per-case band-overlap verdicts — the credibility
@@ -68,7 +101,41 @@ function headlineVerdict(perCase) {
68
101
  return 'WITHIN NOISE — no case moved beyond its confidence band; the skill holds up.';
69
102
  }
70
103
 
71
- function buildDriftReport(a, b, { labelA = 'A', labelB = 'B' } = {}) {
104
+ // A revision pair is a pair in which the SKILL TEXT is the only thing that
105
+ // moved. `diff` was built for release drift, where the model varies and the text
106
+ // is fixed; this inverts it, so the fields that release drift merely warns about
107
+ // become the preconditions of the comparison. Returns the offending field name,
108
+ // or null when the pair is a valid revision pair.
109
+ function revisionPairProblem(a, b) {
110
+ if ((a.run.model_id || '') !== (b.run.model_id || '')) return 'run.model_id';
111
+ if ((a.run.provider || 'anthropic') !== (b.run.provider || 'anthropic')) return 'run.provider';
112
+ if ((a.run.surface || '') !== (b.run.surface || '')) return 'run.surface';
113
+ if ((a.suite.suite_hash || '') !== (b.suite.suite_hash || '')) return 'suite.suite_hash';
114
+ if ((a.skill.content_hash || '') === (b.skill.content_hash || '')) return 'skill.content_hash';
115
+ return null;
116
+ }
117
+
118
+ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' } = {}) {
119
+ const revision = mode === 'revision';
120
+
121
+ // AC-6 (spec 014) — THE PRECONDITION, IN THE PATH THAT EMITS THE VERDICT.
122
+ //
123
+ // A comparison across schema versions assumes the two runs measured the same
124
+ // thing. The assumption is checkable: the no-skill arm contains no skill text,
125
+ // so nothing about a skill or a spec revision can move it. If it fails to
126
+ // reproduce, the comparison has no ground, and Report #006 is what that looks
127
+ // like when it is checked — 3 cells, 0 measured, 3 refused.
128
+ //
129
+ // SCOPED TO CROSS-VERSION PAIRS, which is what AC-6 names. Revision-mode pairs
130
+ // already carry their own register through revisionPairProblem(); widening
131
+ // this to every same-version comparison is a live question, recorded in
132
+ // tasks.md as OPEN-QUESTION-3 rather than decided here.
133
+ const crossVersion = String(a.schema_version || '') !== String(b.schema_version || '');
134
+ let refusal = null;
135
+ if (crossVersion) {
136
+ const pre = baselineReproduces(a, b);
137
+ if (!pre.ok) refusal = { key: pre.key, reason: REFUSAL_REASONS[pre.key](pre) };
138
+ }
72
139
  const aB = withSkillBands(a);
73
140
  const bB = withSkillBands(b);
74
141
  const ids = [...new Set([...Object.keys(aB), ...Object.keys(bB)])];
@@ -79,18 +146,21 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B' } = {}) {
79
146
  // improvement verdicts are never computed from evidence we did not run.
80
147
  const levelOf = (r) => r.verification_level || 'TESTED';
81
148
  const belowTested = [[labelA, a], [labelB, b]].filter(([, r]) => levelOf(r) !== 'TESTED');
82
- const measured = belowTested.length === 0;
149
+ // A REFUSAL IS A RESULT, and it suppresses the verdict exactly as an
150
+ // untested input does: no delta is asserted, and the reason travels with it.
151
+ const measured = belowTested.length === 0 && !refusal;
83
152
 
84
153
  const perCase = ids.map((id) => {
85
154
  const before = aB[id] || null;
86
155
  const after = bB[id] || null;
87
156
  const delta = (before && after) ? round(after.mean - before.mean) : null;
88
- const verdict = !measured ? 'not measured'
157
+ const verdict = refusal ? 'refused'
158
+ : !measured ? 'not measured'
89
159
  : (before && after) ? verdictWithFloor(before, after, delta) : 'n/a';
90
160
  return { id, before, after, delta, verdict };
91
161
  });
92
162
  // Sort worst-first: regressions, then by delta.
93
- const order = { regression: 0, 'within noise': 1, [WITHIN_NOISE_FLOOR]: 1, improvement: 2, 'n/a': 3, 'not measured': 3 };
163
+ const order = { regression: 0, 'within noise': 1, [WITHIN_NOISE_FLOOR]: 1, improvement: 2, 'n/a': 3, 'not measured': 3, refused: 3 };
94
164
  perCase.sort((x, y) => (order[x.verdict] - order[y.verdict]) || ((x.delta || 0) - (y.delta || 0)));
95
165
 
96
166
  const aAgg = aggWithBand(a);
@@ -99,6 +169,32 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B' } = {}) {
99
169
  const regressions = perCase.filter((r) => r.verdict === 'regression');
100
170
 
101
171
  const L = [];
172
+ if (revision) {
173
+ // The axis leads. In a release-drift report the model is what moved and it
174
+ // belongs at the top; here the model is the control and the skill's own text
175
+ // is the finding, so content_hash is the first row a reader meets.
176
+ L.push(`# Revision drift report`);
177
+ L.push('');
178
+ L.push(`**Skill:** ${a.skill.name} \`${a.skill.version}\``);
179
+ L.push('');
180
+ L.push(`**Report type:** revision drift — the skill's own text is the variable under test; the substrate is held fixed.`);
181
+ L.push('');
182
+ L.push(`Left column is the **pinned** revision; right column is the **current** upstream revision.`);
183
+ L.push('');
184
+ L.push(`| | ${labelA} | ${labelB} |`);
185
+ L.push(`|---|---|---|`);
186
+ L.push(`| skill content_hash | \`${short(a.skill.content_hash)}\` | \`${short(b.skill.content_hash)}\` |`);
187
+ L.push(`| model (held) | \`${a.run.model_id}\` | \`${b.run.model_id}\` |`);
188
+ L.push(`| provider (held) | ${a.run.provider || 'anthropic'} | ${b.run.provider || 'anthropic'} |`);
189
+ L.push(`| surface (held) | ${a.run.surface} | ${b.run.surface} |`);
190
+ L.push(`| suite_hash (held) | \`${short(a.suite.suite_hash)}\` | \`${short(b.suite.suite_hash)}\` |`);
191
+ L.push(`| run date (UTC) | ${a.run.date_utc} | ${b.run.date_utc} |`);
192
+ L.push(`| judge samples/case | ${(a.run.judge || {}).samples || 1} | ${(b.run.judge || {}).samples || 1} |`);
193
+ L.push(`| with_skill (mean ± band) | ${bandStr(aAgg)} | ${bandStr(bAgg)} |`);
194
+ L.push(`| baseline score | ${pct(a.comparison.baseline_score)} | ${pct(b.comparison.baseline_score)} |`);
195
+ L.push(`| skill lift (Δ) | ${fmt(a.comparison.delta)} | ${fmt(b.comparison.delta)} |`);
196
+ L.push('');
197
+ } else {
102
198
  L.push(`# Drift report`);
103
199
  L.push('');
104
200
  L.push(`**Skill:** ${a.skill.name} \`${a.skill.version}\``);
@@ -115,19 +211,62 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B' } = {}) {
115
211
  L.push(`| baseline score | ${pct(a.comparison.baseline_score)} | ${pct(b.comparison.baseline_score)} |`);
116
212
  L.push(`| skill lift (Δ) | ${fmt(a.comparison.delta)} | ${fmt(b.comparison.delta)} |`);
117
213
  L.push('');
214
+ }
118
215
 
119
216
  const warnings = [];
120
- if (a.skill.content_hash !== b.skill.content_hash) warnings.push('skill content_hash differs — the skill itself changed between receipts, so drift mixes skill edits with model drift.');
121
- if (a.suite.suite_hash !== b.suite.suite_hash) warnings.push('suite_hash differs — the eval suite changed; per-case comparison may be misleading.');
217
+ // In revision mode the changed skill text is the INDEPENDENT VARIABLE, not
218
+ // contamination, and the held-constant substrate is what makes the comparison
219
+ // valid. The release-drift caveat below says the opposite of both, so it is
220
+ // replaced rather than suppressed: a reader is told what is held, and why a
221
+ // separated band is attributable to the revision.
222
+ if (revision) {
223
+ warnings.push(`model \`${a.run.model_id}\`, provider ${a.run.provider || 'anthropic'}, surface ${a.run.surface} and suite_hash \`${short(a.suite.suite_hash)}\` are held constant across both receipts — the skill text is the only variable under test, so a band-separated move above the ${EFFECT_FLOOR} floor is attributable to the revision.`);
224
+ }
225
+ if (!revision && a.skill.content_hash !== b.skill.content_hash) warnings.push('skill content_hash differs — the skill itself changed between receipts, so drift mixes skill edits with model drift.');
226
+ if (!revision && a.suite.suite_hash !== b.suite.suite_hash) warnings.push('suite_hash differs — the eval suite changed; per-case comparison may be misleading.');
122
227
  if (a.skill.name !== b.skill.name) warnings.push(`different skills (${a.skill.name} vs ${b.skill.name}) — comparison is not meaningful.`);
123
228
  if ((a.run.judge || {}).samples <= 1 || (b.run.judge || {}).samples <= 1) warnings.push('one or both receipts are single-sample (no bands) — non-overlap can only be trusted when both sides are sampled.');
124
- if (!measured) warnings.push(`verdicts NOT computed — ${belowTested.map(([l, r]) => `${l} is ${levelOf(r)}${r.run && r.run.source ? ` (${r.run.source})` : ''}`).join('; ')}. Drift verdicts require TESTED receipts on both sides; declared numbers are shown as context only (see /interop.html).`);
229
+ // WHAT SUPPRESSED THE VERDICT IS SAID, AND IT IS SAID CORRECTLY (spec 016 AC-3
230
+ // and AC-4, closing F-015-B).
231
+ //
232
+ // TWO DIFFERENT THINGS can suppress a verdict, and this line used to describe
233
+ // only one of them. A REFUSAL — the baseline-reproduction precondition — set
234
+ // `measured` false and then the caveat rendered `belowTested`, which on a
235
+ // refused pair is EMPTY: the page read `verdicts NOT computed — .` under a
236
+ // headline claiming `0 receipt(s) below TESTED`, on a pair where both receipts
237
+ // were TESTED. An empty reason and a false statement, in the artifact a report's
238
+ // verdicts come from. The refusal's own cause-honest reason was computed into
239
+ // `refusal.reason` and thrown away by the renderer, so nothing a reader could
240
+ // see said why the comparison had stopped.
241
+ //
242
+ // Each half is proved load-bearing by a mutation that patches THIS source in a
243
+ // disposable copy. An earlier revision guarded them with const-true flags and
244
+ // this sentence described those; the flags were removed because a
245
+ // defect-restoring branch resident in the published tree is a hazard, and the
246
+ // sentence outlived them by one commit.
247
+ if (refusal) {
248
+ warnings.push(`verdicts NOT computed — the comparison was REFUSED before any verdict was formed: ${refusal.reason}`);
249
+ }
250
+ if (belowTested.length) {
251
+ warnings.push(`verdicts NOT computed — ${belowTested.map(([l, r]) => `${l} is ${levelOf(r)}${r.run && r.run.source ? ` (${r.run.source})` : ''}`).join('; ')}. Drift verdicts require TESTED receipts on both sides; declared numbers are shown as context only (see /interop.html).`);
252
+ }
125
253
  // Cross-provider / cross-surface disclosure (Phase 6). A comparison across
126
254
  // providers is a skill-DURABILITY comparison across substrates, not model drift
127
255
  // over time; across surfaces, sampling control differs. Both are flagged so a
128
256
  // reader never mistakes one for the other (see docs/neutrality.html).
129
- if ((a.run.provider || 'anthropic') !== (b.run.provider || 'anthropic')) warnings.push(`different providers (${a.run.provider || 'anthropic'} vs ${b.run.provider || 'anthropic'}) — this is a cross-substrate durability comparison, not model drift over time; read the delta, not absolute scores (see the neutrality policy).`);
130
- if (a.run.surface !== b.run.surface) warnings.push(`different surfaces (${a.run.surface} vs ${b.run.surface}) — sampling control differs between surfaces; compare with care.`);
257
+ if (!revision && (a.run.provider || 'anthropic') !== (b.run.provider || 'anthropic')) warnings.push(`different providers (${a.run.provider || 'anthropic'} vs ${b.run.provider || 'anthropic'}) — this is a cross-substrate durability comparison, not model drift over time; read the delta, not absolute scores (see the neutrality policy).`);
258
+ if (!revision && a.run.surface !== b.run.surface) warnings.push(`different surfaces (${a.run.surface} vs ${b.run.surface}) — sampling control differs between surfaces; compare with care.`);
259
+ // The legend for the band markers, printed whenever either side carries one.
260
+ // A two-letter marker a reader cannot decode is worse than no marker.
261
+ // GATED ON WHETHER A LABEL IS ACTUALLY RENDERED, not on whether the data
262
+ // carries a source. A legend explains markers on the page; if `bandStr` emits
263
+ // none, the legend is describing something the reader cannot see. Asked of
264
+ // `bandStr` itself rather than recomputed, so the two cannot disagree.
265
+ const anyLabelRendered = [...Object.values(aB), ...Object.values(bB)]
266
+ .some((x) => x && /\([a-z]+\)\s*$/.test(bandStr(x)));
267
+ if (anyLabelRendered) {
268
+ warnings.push('band provenance: `(generation)` is an ACROSS-DRAW spread — receipt spec v0.5, n generation draws per arm. `(legacy)` is a JUDGE-SAMPLE spread over a single generation, which is what v0.4 and earlier recorded. They are different statistics. The comparison is the only one the older receipt admits, and it is not like for like.');
269
+ }
131
270
  if (warnings.length) {
132
271
  L.push('> **⚠ Caveats**');
133
272
  for (const w of warnings) L.push(`> - ${w}`);
@@ -139,10 +278,23 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B' } = {}) {
139
278
  L.push(`## Headline`);
140
279
  L.push('');
141
280
  if (!measured) {
142
- L.push(`**NOT MEASURED — ${belowTested.length} receipt(s) below TESTED. Drift verdicts require TESTED receipts on both sides; the declared numbers above are context, not band-verified evidence.**`);
281
+ // THE HEADLINE NAMES THE CAUSE THAT ACTUALLY APPLIES. `belowTested.length`
282
+ // was printed unconditionally, so a refused pair got "0 receipt(s) below
283
+ // TESTED" — a statement measurably false of the receipts it was given.
284
+ if (refusal) {
285
+ // The reason is a complete sentence and already ends by saying no verdict
286
+ // is asserted; prefixing that again produced "REFUSED — no verdict is
287
+ // asserted. the baseline arm did not reproduce…" — a duplicated clause and
288
+ // a lower-case sentence start, in the artifact a published report quotes.
289
+ L.push(`**REFUSED — ${refusal.reason}**`);
290
+ } else if (belowTested.length) {
291
+ L.push(`**NOT MEASURED — ${belowTested.length} receipt(s) below TESTED. Drift verdicts require TESTED receipts on both sides; the declared numbers above are context, not band-verified evidence.**`);
292
+ } else {
293
+ L.push('**NOT MEASURED — no verdict is asserted.**');
294
+ }
143
295
  L.push('');
144
296
  } else {
145
- L.push(`**${headlineVerdict(perCase)}**`);
297
+ L.push(`**${revision ? revisionHeadline(perCase) : headlineVerdict(perCase)}**`);
146
298
  L.push('');
147
299
  L.push(`with_skill mean moved ${fmt(headlineDelta)} (${bandStr(aAgg)} → ${bandStr(bAgg)}; band = suite dispersion). Per-case band-overlap verdicts: ${regressions.length} regression(s), ${perCase.filter((r) => r.verdict === 'improvement').length} improvement(s), ${nWithin} within noise${nFloor ? ` (${nFloor} of them band-separated but below the ${EFFECT_FLOOR} effect floor)` : ''}.`);
148
300
  L.push('');
@@ -167,7 +319,12 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B' } = {}) {
167
319
  L.push('');
168
320
  }
169
321
 
170
- return { markdown: L.join('\n'), perCase, headlineDelta, regressions };
322
+ return {
323
+ markdown: L.join('\n'), perCase, headlineDelta, regressions,
324
+ refused: !!refusal,
325
+ refusal_reason: refusal ? refusal.reason : null,
326
+ refusal_key: refusal ? refusal.key : null,
327
+ };
171
328
  }
172
329
 
173
- module.exports = { buildDriftReport, withSkillBands };
330
+ module.exports = { buildDriftReport, withSkillBands, revisionPairProblem };
package/lib/hygiene.js ADDED
@@ -0,0 +1,113 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ //
3
+ // The single definition of what counts as a leak.
4
+ //
5
+ // Two callers scan for the same classes of secret: `tests/gate.js`, which walks
6
+ // the working tree at gate time, and `scripts/merge-check.js`, which walks the
7
+ // candidate tree read out of the commit before permitting a merge. They must
8
+ // agree. Two copies of a deny-list drift, and the copy that quietly stops
9
+ // matching still reads as protection — the same argument the acknowledgment
10
+ // deny-list bite test already makes, applied to the scanner itself.
11
+ //
12
+ // Backlog B-8 records four approval records that reached `dev` carrying a host
13
+ // path. The recorded cause — "the scan walks tracked files" — is false, and this
14
+ // module's existence does not depend on it: `listFiles()` in the repo gate has
15
+ // always enumerated untracked files too, and a planted path in an untracked
16
+ // evidence record does take the repo gate red. The real window is VANTAGE. An
17
+ // approval session measures in a disposable copy made before it writes its
18
+ // record into the shared checkout, so the record is simply not in the tree that
19
+ // was scanned. Scanning the working tree harder cannot close that. Scanning the
20
+ // commit that is about to merge does, and that is what merge-check now uses this
21
+ // for.
22
+ //
23
+ // Patterns are written so a regex literal cannot match its own text: a
24
+ // metacharacter or a character class follows each fixed prefix.
25
+
26
+ const dec = (b64) => JSON.parse(Buffer.from(b64, 'base64').toString('utf8'));
27
+
28
+ // Build-box host + private IP, base64 so the plaintext never lands in a file.
29
+ const HOSTIP = dec('WyJpcC0xNzItMzEtNDAtMTU5LmFwLXNvdXRoZWFzdC0xLmNvbXB1dGUuaW50ZXJuYWwiLCIxNzIuMzEuNDAuMTU5Il0=');
30
+
31
+ const EMAIL_ALLOW = new Set(['example.com', 'example.org', 'driftproofhq.com']);
32
+
33
+ const PATTERNS = [
34
+ { name: 'home-path', re: /\/home\/[a-z0-9_-]+\/|\/Users\/[A-Za-z0-9_-]+\// },
35
+ { name: 'ec2-internal-host', re: /ip-\d+-\d+-\d+-\d+\.[a-z0-9.-]*compute\.(internal|amazonaws\.com)/i },
36
+ { name: 'private-ip', re: /\b(10\.\d{1,3}\.\d{1,3}\.\d{1,3}|172\.(1[6-9]|2\d|3[01])\.\d{1,3}\.\d{1,3}|192\.168\.\d{1,3}\.\d{1,3})\b/ },
37
+ { name: 'anthropic-key', re: /sk-ant-[A-Za-z0-9_-]{8,}/ },
38
+ { name: 'github-token', re: /gh[pousr]_[A-Za-z0-9]{20,}/ },
39
+ { name: 'aws-key', re: /\bAKIA[0-9A-Z]{16}\b/ },
40
+ { name: 'private-key-block', re: /-----BEGIN [A-Z ]*PRIVATE KEY-----/ },
41
+ { name: 'env-secret-assignment', re: /\b(ANTHROPIC_API_KEY|AWS_SECRET_ACCESS_KEY|OPENAI_API_KEY)\s*=\s*\S+/ },
42
+ // Codex subscription auth material (~/.codex/auth.json contents) must NEVER
43
+ // land in a committed file: the id_token/access_token are JWTs, and an OpenAI
44
+ // secret key is sk-proj-/sk-svcacct-/sk-admin-. We ban the CONTENTS (tokens),
45
+ // not the documented path string (`~/.codex/auth.json` is referenced in help
46
+ // text and docs by design).
47
+ { name: 'jwt-token', re: /\beyJ[A-Za-z0-9_=-]{10,}\.eyJ[A-Za-z0-9_=-]{10,}\.[A-Za-z0-9_=-]{6,}/ },
48
+ { name: 'openai-secret-key', re: /\bsk-(proj|svcacct|admin)-[A-Za-z0-9_-]{20,}/ },
49
+ ];
50
+
51
+ // A path is a hit on its own name, with no content read: a committed .env file
52
+ // is a leak whatever it happens to contain.
53
+ function scanPath(rel) {
54
+ return /(^|\/)\.env(\.|$)/.test(rel) ? [{ file: rel, kind: 'env-file' }] : [];
55
+ }
56
+
57
+ function scanContent(rel, content) {
58
+ const hits = [];
59
+ for (const p of PATTERNS) {
60
+ const m = content.match(p.re);
61
+ if (m) hits.push({ file: rel, kind: p.name, sample: m[0].slice(0, 24) });
62
+ }
63
+ for (const hv of HOSTIP) if (content.includes(hv)) hits.push({ file: rel, kind: 'build-host-or-ip' });
64
+ const emails = content.match(/\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b/g) || [];
65
+ for (const e of emails) {
66
+ const dom = e.split('@')[1].toLowerCase();
67
+ if (!EMAIL_ALLOW.has(dom) && !dom.endsWith('.example')) hits.push({ file: rel, kind: 'email', sample: e });
68
+ }
69
+ return hits;
70
+ }
71
+
72
+ // scanFiles(files, read, readLink) — `read(rel)` returns the file's text, or
73
+ // throws/returns null for anything unreadable (binary, deleted, a symlink to
74
+ // nowhere), which is skipped exactly as the gate has always skipped it.
75
+ //
76
+ // `readLink(rel)` is optional and returns a symbolic link's TARGET PATH as a
77
+ // string, or null for an entry that is not a link. Where it is supplied, that
78
+ // target is scanned as the entry's content through `scanContent`, so every
79
+ // pattern above applies to it with no second deny-list to drift (spec 011,
80
+ // AC-2). A link's target is a path, and a path is exactly the class of leak the
81
+ // `home-path` pattern exists to catch: P-1 shipped `node_modules ->
82
+ // /home/<user>/... ` into a public tree past a scan that read the link as an
83
+ // unreadable file and skipped it.
84
+ //
85
+ // The argument is optional so a caller supplying nothing behaves exactly as
86
+ // before. That compatibility is deliberate, and it is also the standing risk: a
87
+ // caller added later is blind by default. The repo gate asserts that every call
88
+ // site supplies a reader, with `scripts/merge-check.js` the one exception — its
89
+ // reader is `git show <ref>:<path>`, and git stores a link's blob as its target
90
+ // string, so the target already arrives as content there.
91
+ function scanFiles(files, read, readLink) {
92
+ const hits = [];
93
+ for (const rel of files) {
94
+ hits.push(...scanPath(rel));
95
+ // The link branch is a single condition on purpose: it is the mutation seam
96
+ // the spec-011 gate neutralises to prove the scanner goes blind without it.
97
+ if (typeof readLink === 'function') {
98
+ let target;
99
+ try { target = readLink(rel); } catch (_e) { target = null; }
100
+ if (typeof target === 'string') {
101
+ hits.push(...scanContent(rel, target));
102
+ continue;
103
+ }
104
+ }
105
+ let c;
106
+ try { c = read(rel); } catch (_e) { continue; }
107
+ if (typeof c !== 'string') continue;
108
+ hits.push(...scanContent(rel, c));
109
+ }
110
+ return hits;
111
+ }
112
+
113
+ module.exports = { PATTERNS, EMAIL_ALLOW, HOSTIP, scanPath, scanContent, scanFiles };
package/lib/judge.js CHANGED
@@ -99,7 +99,12 @@ async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature
99
99
  // { samples:[scores], mean, stddev, reason, judge_settings, model_id, rubric_hash }
100
100
  // `mean` ± `stddev` is the per-case confidence band (raw spread of the N scores)
101
101
  // used by the borderline-outcome rule and per-case drift band-overlap logic.
102
- async function gradeSamples({ task, response, rubric, model, samples = 5, timeoutMs = 120000 }) {
102
+ // NO DEFAULT TIMEOUT HERE (spec 017 AC-2). This defaulted to 120000, which
103
+ // outranked the per-surface policy exactly as lib/run.js's literal did — so the
104
+ // JUDGE calls timed out on the api policy while running on a CLI surface, which
105
+ // is the second shadowing site and the one #007's prep session had not found.
106
+ // Passing `undefined` through lets lib/provider.js resolve the declared policy.
107
+ async function gradeSamples({ task, response, rubric, model, samples = 5, timeoutMs }) {
103
108
  const settings = judgeSettings(samples, model);
104
109
  const scores = [];
105
110
  const reasons = [];
package/lib/receipt.js CHANGED
@@ -15,7 +15,12 @@ const SCHEMA_FILES = {
15
15
  '0.2': 'receipt.v0.2.schema.json',
16
16
  '0.3': 'receipt.v0.3.schema.json',
17
17
  '0.3.1': 'receipt.v0.3.1.schema.json',
18
- '0.4': 'receipt.schema.json',
18
+ // v0.4 moved from the unversioned filename to a version-pinned one when v0.5
19
+ // took the current pointer. Without this the archive would lose v0.4: every
20
+ // published v0.4 receipt asserts conformance by NUMBER, and the number has to
21
+ // keep resolving to the schema it meant (AC-10).
22
+ '0.4': 'receipt.v0.4.schema.json',
23
+ '0.5': 'receipt.schema.json',
19
24
  };
20
25
 
21
26
  const _validators = {};
@@ -84,15 +89,70 @@ function aggregate(caseResults) {
84
89
  function buildReceipt({ skill, suite, run, cases, economics = null, verificationLevel = 'TESTED', editorialReviews = null }) {
85
90
  // v0.3.1: cases marked failed_timeout are recorded in results.cases but EXCLUDED
86
91
  // from aggregates — a band is never fabricated from a case that did not complete.
87
- const okCases = cases.filter((c) => c.case_status !== 'failed_timeout');
88
- const failedCount = cases.length - okCases.length;
92
+ //
93
+ // PAIRWISE, NOT PER ARM (spec 017 AC-5, AC-6). This filtered a FLAT list and
94
+ // then split by mode, so a case whose baseline failed kept its with_skill arm:
95
+ // #007's cell 3 recorded with_skill case_count 7 against baseline 6, and its
96
+ // headline `delta` of +0.116 was a 7-case mean minus a 6-case mean.
97
+ //
98
+ // `comparison.delta` is PAIRED BY CONSTRUCTION — one suite, measured twice —
99
+ // and a paired statistic computed over unequal sets is not the statistic it
100
+ // names. So an arm that cannot be measured removes its CASE from both sides.
101
+ //
102
+ // Exclusion, not refusal, and the reason is recorded rather than argued: an
103
+ // aggregate is a summary statistic, not a verdict, so the "refuse rather than
104
+ // assert" rule that governs verdicts does not reach it; and refusing the whole
105
+ // aggregate would discard thirteen sound arms because one failed. What the
106
+ // aggregate owes a reader instead is that it says what it covered, which
107
+ // `excluded_cases` provides.
108
+ //
109
+ // The excluded case STAYS in `results.cases`. It is removed from the mean, not
110
+ // from the record — deleting the evidence of a failure is a different and worse
111
+ // defect than averaging over it.
112
+ const armUnusable = (c) => c.case_status === 'failed_timeout'
113
+ || (c.mean == null && c.score == null);
114
+ const excludedIds = new Map();
115
+ for (const c of cases) {
116
+ if (!armUnusable(c)) continue;
117
+ if (excludedIds.has(c.id)) { excludedIds.get(c.id).modes.push(c.mode); continue; }
118
+ excludedIds.set(c.id, {
119
+ id: c.id,
120
+ modes: [c.mode],
121
+ reason: c.reason
122
+ || (c.generation && c.generation.stopping_reason === 'unmeasured_exhausted'
123
+ ? 'every generation draw was unmeasured'
124
+ : 'the arm has no measured result'),
125
+ });
126
+ }
127
+ const okCases = cases.filter((c) => !excludedIds.has(c.id));
89
128
  const withSkill = okCases.filter((c) => c.mode === 'with_skill');
90
129
  const baseline = okCases.filter((c) => c.mode === 'baseline');
130
+ const excludedCases = [...excludedIds.values()];
131
+ // `run.failed_case_count` COUNTS CASES, and is computed FROM the list it
132
+ // summarises rather than alongside it, so the two fields cannot disagree.
133
+ //
134
+ // It was `cases.length - okCases.length`, which counted ROWS. That was the
135
+ // same number while exclusion was per arm; once AC-5 made exclusion pairwise
136
+ // the subtraction removed BOTH arms of every excluded case, so one failed arm
137
+ // reported 2. Measured on the archive before this fix: #006's writing-plans
138
+ // receipt recomputed to 4 against the 2 it records, and #007's to 2 against 1
139
+ // — two published figures doubled by a change that never named this field.
140
+ //
141
+ // The field's name says cases and the aggregates exclude by case, so the
142
+ // count is the length of `results.aggregates.excluded_cases` and nothing
143
+ // else. A case whose BOTH arms failed is one exclusion and counts once.
144
+ const failedCount = excludedCases.length;
91
145
  const aggWith = aggregate(withSkill);
92
146
  const aggBase = aggregate(baseline);
93
147
 
94
148
  const receipt = {
95
149
  schema_version: RECEIPT_SCHEMA_VERSION,
150
+ // v0.5 CAPABILITY FLAG (F-014-F). DERIVED FROM THE CASES THEMSELVES, never
151
+ // from a caller's argument: the exposure this closes is a receipt asserting
152
+ // something it does not carry, and a flag taken on trust from the caller
153
+ // would be the same defect with an extra step. Absent when no case carries a
154
+ // draw set, which is what keeps every legacy and imported receipt valid.
155
+ ...(cases.some((c) => c && c.generation) ? { generation_sampled: true } : {}),
96
156
  skill: {
97
157
  name: skill.name,
98
158
  version: skill.version,
@@ -102,6 +162,12 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
102
162
  format: suite.format,
103
163
  suite_hash: suite.suiteHash,
104
164
  case_count: suite.caseCount,
165
+ // v0.5: the per-suite canary. THIS is the canonical assembly — the runner
166
+ // built the field and this function dropped it, so the live smoke emitted
167
+ // `canary: undefined` and the schema, which makes it optional, said
168
+ // nothing. Caught by reading the receipt a real run produced, not by a
169
+ // gate; the assertion that would have caught it is added with the fix.
170
+ ...(suite.canary ? { canary: suite.canary } : {}),
105
171
  },
106
172
  run: {
107
173
  model_id: run.model_id,
@@ -120,7 +186,13 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
120
186
  },
121
187
  results: {
122
188
  cases,
123
- aggregates: { with_skill: aggWith, baseline: aggBase },
189
+ aggregates: {
190
+ with_skill: aggWith,
191
+ baseline: aggBase,
192
+ // Present only when something was excluded, so a clean run's receipt is
193
+ // unchanged and the archive does not acquire an empty field.
194
+ ...(excludedCases.length ? { excluded_cases: excludedCases } : {}),
195
+ },
124
196
  },
125
197
  comparison: {
126
198
  with_skill_score: aggWith.mean_score,