driftproof 0.5.0 → 0.7.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +104 -29
- package/bin/driftproof +46 -12
- package/config.js +49 -5
- package/lib/canary.js +27 -0
- package/lib/cost.js +20 -2
- package/lib/diff.js +172 -15
- package/lib/hygiene.js +113 -0
- package/lib/judge.js +6 -1
- package/lib/receipt.js +76 -4
- package/lib/reuse.js +134 -0
- package/lib/revision.js +169 -0
- package/lib/run.js +165 -45
- package/lib/sampling.js +110 -0
- package/lib/stats.js +16 -1
- package/lib/value.js +27 -2
- package/package.json +2 -2
- package/spec/RECEIPT.md +102 -3
- package/spec/receipt.schema.json +360 -3
- package/spec/receipt.v0.4.schema.json +961 -0
package/lib/diff.js
CHANGED
|
@@ -3,6 +3,8 @@
|
|
|
3
3
|
|
|
4
4
|
const { bandVerdict, round } = require('./stats');
|
|
5
5
|
const { EFFECT_FLOOR } = require('../config');
|
|
6
|
+
const { revisionHeadline } = require('./revision');
|
|
7
|
+
const { baselineReproduces, REFUSAL_REASONS, bandOf } = require('./reuse');
|
|
6
8
|
|
|
7
9
|
// Practical-significance gate applied ON TOP of band separation. bandVerdict()
|
|
8
10
|
// stays a pure geometry test (kept that way so its unit checks are unambiguous);
|
|
@@ -31,11 +33,23 @@ function isWithinNoise(v) { return v === 'within noise' || v === WITHIN_NOISE_FL
|
|
|
31
33
|
|
|
32
34
|
// Map case-id → { mean, stddev } for a receipt's with_skill cases. Falls back to
|
|
33
35
|
// score/0 for v0.1 receipts that have no per-case band.
|
|
36
|
+
// NO RENDERING SWITCHES. An earlier revision carried two const-true flags so each
|
|
37
|
+
// half of F-015-B's fix could be removed in a probe — and the dead branch of one
|
|
38
|
+
// of them contained, verbatim, the false headline AC-4 exists to forbid. Nothing
|
|
39
|
+
// could fire it, and it was still a defect-restoring branch living in the tree
|
|
40
|
+
// that gets published. The mutation probes patch this source in a disposable copy.
|
|
41
|
+
|
|
34
42
|
function withSkillBands(receipt) {
|
|
35
43
|
const out = {};
|
|
36
44
|
for (const c of receipt.results.cases) {
|
|
37
45
|
if (c.mode !== 'with_skill') continue;
|
|
38
|
-
|
|
46
|
+
// ONE BAND DEFINITION (spec 016 AC-1). This built its own from the v0.4-shaped
|
|
47
|
+
// fields, which worked — and that is the point: three copies existed, two
|
|
48
|
+
// reading the legacy shape and one reading only v0.5, and the one that read
|
|
49
|
+
// only v0.5 was the one a cross-version control depended on (F-015-C). Routing
|
|
50
|
+
// every comparison path through `bandOf` means a future shape is added once.
|
|
51
|
+
const b = bandOf(c);
|
|
52
|
+
if (b) out[c.id] = { mean: b.mean, stddev: b.sd, source: b.source };
|
|
39
53
|
}
|
|
40
54
|
return out;
|
|
41
55
|
}
|
|
@@ -50,7 +64,26 @@ function aggWithBand(receipt) {
|
|
|
50
64
|
|
|
51
65
|
function fmt(n) { return n == null ? 'n/a' : (n >= 0 ? '+' : '') + n.toFixed(3); }
|
|
52
66
|
function pct(n) { return n == null ? 'n/a' : n.toFixed(3); }
|
|
53
|
-
|
|
67
|
+
// THE BAND SAYS WHICH BAND IT IS (spec 017 AC-7).
|
|
68
|
+
//
|
|
69
|
+
// `bandOf` computes `source: 'legacy' | 'generation'` and carries it onto every
|
|
70
|
+
// band; nothing rendered it, so a reader comparing an archived receipt with a
|
|
71
|
+
// v0.5 one was comparing a JUDGE-SAMPLE spread against an ACROSS-DRAW spread
|
|
72
|
+
// with nothing on the page saying so. They are different statistics over
|
|
73
|
+
// different things, and the comparison is still the only one v0.4 admits — which
|
|
74
|
+
// is exactly why the page has to name them rather than leave them to look alike.
|
|
75
|
+
//
|
|
76
|
+
// THE MARKER IS THE RECEIPT'S OWN WORD — `legacy` or `generation`, exactly as
|
|
77
|
+
// `bandOf` records it — rather than a prettier synonym. A reader who greps the
|
|
78
|
+
// page for what a receipt says should find the same token; a rendering that
|
|
79
|
+
// renames the thing it is disclosing has disclosed a different thing. Omitted
|
|
80
|
+
// when a band carries no source (an aggregate band is computed from case means,
|
|
81
|
+
// not from one case's draws).
|
|
82
|
+
|
|
83
|
+
function bandStr(x) {
|
|
84
|
+
const label = x && x.source;
|
|
85
|
+
return `${x.mean.toFixed(3)} ± ${x.stddev.toFixed(3)}${label ? ` (${label})` : ''}`;
|
|
86
|
+
}
|
|
54
87
|
function short(h) { return h ? String(h).slice(0, 12) : 'n/a'; }
|
|
55
88
|
|
|
56
89
|
// The headline is a SUMMARY of the per-case band-overlap verdicts — the credibility
|
|
@@ -68,7 +101,41 @@ function headlineVerdict(perCase) {
|
|
|
68
101
|
return 'WITHIN NOISE — no case moved beyond its confidence band; the skill holds up.';
|
|
69
102
|
}
|
|
70
103
|
|
|
71
|
-
|
|
104
|
+
// A revision pair is a pair in which the SKILL TEXT is the only thing that
|
|
105
|
+
// moved. `diff` was built for release drift, where the model varies and the text
|
|
106
|
+
// is fixed; this inverts it, so the fields that release drift merely warns about
|
|
107
|
+
// become the preconditions of the comparison. Returns the offending field name,
|
|
108
|
+
// or null when the pair is a valid revision pair.
|
|
109
|
+
function revisionPairProblem(a, b) {
|
|
110
|
+
if ((a.run.model_id || '') !== (b.run.model_id || '')) return 'run.model_id';
|
|
111
|
+
if ((a.run.provider || 'anthropic') !== (b.run.provider || 'anthropic')) return 'run.provider';
|
|
112
|
+
if ((a.run.surface || '') !== (b.run.surface || '')) return 'run.surface';
|
|
113
|
+
if ((a.suite.suite_hash || '') !== (b.suite.suite_hash || '')) return 'suite.suite_hash';
|
|
114
|
+
if ((a.skill.content_hash || '') === (b.skill.content_hash || '')) return 'skill.content_hash';
|
|
115
|
+
return null;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' } = {}) {
|
|
119
|
+
const revision = mode === 'revision';
|
|
120
|
+
|
|
121
|
+
// AC-6 (spec 014) — THE PRECONDITION, IN THE PATH THAT EMITS THE VERDICT.
|
|
122
|
+
//
|
|
123
|
+
// A comparison across schema versions assumes the two runs measured the same
|
|
124
|
+
// thing. The assumption is checkable: the no-skill arm contains no skill text,
|
|
125
|
+
// so nothing about a skill or a spec revision can move it. If it fails to
|
|
126
|
+
// reproduce, the comparison has no ground, and Report #006 is what that looks
|
|
127
|
+
// like when it is checked — 3 cells, 0 measured, 3 refused.
|
|
128
|
+
//
|
|
129
|
+
// SCOPED TO CROSS-VERSION PAIRS, which is what AC-6 names. Revision-mode pairs
|
|
130
|
+
// already carry their own register through revisionPairProblem(); widening
|
|
131
|
+
// this to every same-version comparison is a live question, recorded in
|
|
132
|
+
// tasks.md as OPEN-QUESTION-3 rather than decided here.
|
|
133
|
+
const crossVersion = String(a.schema_version || '') !== String(b.schema_version || '');
|
|
134
|
+
let refusal = null;
|
|
135
|
+
if (crossVersion) {
|
|
136
|
+
const pre = baselineReproduces(a, b);
|
|
137
|
+
if (!pre.ok) refusal = { key: pre.key, reason: REFUSAL_REASONS[pre.key](pre) };
|
|
138
|
+
}
|
|
72
139
|
const aB = withSkillBands(a);
|
|
73
140
|
const bB = withSkillBands(b);
|
|
74
141
|
const ids = [...new Set([...Object.keys(aB), ...Object.keys(bB)])];
|
|
@@ -79,18 +146,21 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B' } = {}) {
|
|
|
79
146
|
// improvement verdicts are never computed from evidence we did not run.
|
|
80
147
|
const levelOf = (r) => r.verification_level || 'TESTED';
|
|
81
148
|
const belowTested = [[labelA, a], [labelB, b]].filter(([, r]) => levelOf(r) !== 'TESTED');
|
|
82
|
-
|
|
149
|
+
// A REFUSAL IS A RESULT, and it suppresses the verdict exactly as an
|
|
150
|
+
// untested input does: no delta is asserted, and the reason travels with it.
|
|
151
|
+
const measured = belowTested.length === 0 && !refusal;
|
|
83
152
|
|
|
84
153
|
const perCase = ids.map((id) => {
|
|
85
154
|
const before = aB[id] || null;
|
|
86
155
|
const after = bB[id] || null;
|
|
87
156
|
const delta = (before && after) ? round(after.mean - before.mean) : null;
|
|
88
|
-
const verdict =
|
|
157
|
+
const verdict = refusal ? 'refused'
|
|
158
|
+
: !measured ? 'not measured'
|
|
89
159
|
: (before && after) ? verdictWithFloor(before, after, delta) : 'n/a';
|
|
90
160
|
return { id, before, after, delta, verdict };
|
|
91
161
|
});
|
|
92
162
|
// Sort worst-first: regressions, then by delta.
|
|
93
|
-
const order = { regression: 0, 'within noise': 1, [WITHIN_NOISE_FLOOR]: 1, improvement: 2, 'n/a': 3, 'not measured': 3 };
|
|
163
|
+
const order = { regression: 0, 'within noise': 1, [WITHIN_NOISE_FLOOR]: 1, improvement: 2, 'n/a': 3, 'not measured': 3, refused: 3 };
|
|
94
164
|
perCase.sort((x, y) => (order[x.verdict] - order[y.verdict]) || ((x.delta || 0) - (y.delta || 0)));
|
|
95
165
|
|
|
96
166
|
const aAgg = aggWithBand(a);
|
|
@@ -99,6 +169,32 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B' } = {}) {
|
|
|
99
169
|
const regressions = perCase.filter((r) => r.verdict === 'regression');
|
|
100
170
|
|
|
101
171
|
const L = [];
|
|
172
|
+
if (revision) {
|
|
173
|
+
// The axis leads. In a release-drift report the model is what moved and it
|
|
174
|
+
// belongs at the top; here the model is the control and the skill's own text
|
|
175
|
+
// is the finding, so content_hash is the first row a reader meets.
|
|
176
|
+
L.push(`# Revision drift report`);
|
|
177
|
+
L.push('');
|
|
178
|
+
L.push(`**Skill:** ${a.skill.name} \`${a.skill.version}\``);
|
|
179
|
+
L.push('');
|
|
180
|
+
L.push(`**Report type:** revision drift — the skill's own text is the variable under test; the substrate is held fixed.`);
|
|
181
|
+
L.push('');
|
|
182
|
+
L.push(`Left column is the **pinned** revision; right column is the **current** upstream revision.`);
|
|
183
|
+
L.push('');
|
|
184
|
+
L.push(`| | ${labelA} | ${labelB} |`);
|
|
185
|
+
L.push(`|---|---|---|`);
|
|
186
|
+
L.push(`| skill content_hash | \`${short(a.skill.content_hash)}\` | \`${short(b.skill.content_hash)}\` |`);
|
|
187
|
+
L.push(`| model (held) | \`${a.run.model_id}\` | \`${b.run.model_id}\` |`);
|
|
188
|
+
L.push(`| provider (held) | ${a.run.provider || 'anthropic'} | ${b.run.provider || 'anthropic'} |`);
|
|
189
|
+
L.push(`| surface (held) | ${a.run.surface} | ${b.run.surface} |`);
|
|
190
|
+
L.push(`| suite_hash (held) | \`${short(a.suite.suite_hash)}\` | \`${short(b.suite.suite_hash)}\` |`);
|
|
191
|
+
L.push(`| run date (UTC) | ${a.run.date_utc} | ${b.run.date_utc} |`);
|
|
192
|
+
L.push(`| judge samples/case | ${(a.run.judge || {}).samples || 1} | ${(b.run.judge || {}).samples || 1} |`);
|
|
193
|
+
L.push(`| with_skill (mean ± band) | ${bandStr(aAgg)} | ${bandStr(bAgg)} |`);
|
|
194
|
+
L.push(`| baseline score | ${pct(a.comparison.baseline_score)} | ${pct(b.comparison.baseline_score)} |`);
|
|
195
|
+
L.push(`| skill lift (Δ) | ${fmt(a.comparison.delta)} | ${fmt(b.comparison.delta)} |`);
|
|
196
|
+
L.push('');
|
|
197
|
+
} else {
|
|
102
198
|
L.push(`# Drift report`);
|
|
103
199
|
L.push('');
|
|
104
200
|
L.push(`**Skill:** ${a.skill.name} \`${a.skill.version}\``);
|
|
@@ -115,19 +211,62 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B' } = {}) {
|
|
|
115
211
|
L.push(`| baseline score | ${pct(a.comparison.baseline_score)} | ${pct(b.comparison.baseline_score)} |`);
|
|
116
212
|
L.push(`| skill lift (Δ) | ${fmt(a.comparison.delta)} | ${fmt(b.comparison.delta)} |`);
|
|
117
213
|
L.push('');
|
|
214
|
+
}
|
|
118
215
|
|
|
119
216
|
const warnings = [];
|
|
120
|
-
|
|
121
|
-
|
|
217
|
+
// In revision mode the changed skill text is the INDEPENDENT VARIABLE, not
|
|
218
|
+
// contamination, and the held-constant substrate is what makes the comparison
|
|
219
|
+
// valid. The release-drift caveat below says the opposite of both, so it is
|
|
220
|
+
// replaced rather than suppressed: a reader is told what is held, and why a
|
|
221
|
+
// separated band is attributable to the revision.
|
|
222
|
+
if (revision) {
|
|
223
|
+
warnings.push(`model \`${a.run.model_id}\`, provider ${a.run.provider || 'anthropic'}, surface ${a.run.surface} and suite_hash \`${short(a.suite.suite_hash)}\` are held constant across both receipts — the skill text is the only variable under test, so a band-separated move above the ${EFFECT_FLOOR} floor is attributable to the revision.`);
|
|
224
|
+
}
|
|
225
|
+
if (!revision && a.skill.content_hash !== b.skill.content_hash) warnings.push('skill content_hash differs — the skill itself changed between receipts, so drift mixes skill edits with model drift.');
|
|
226
|
+
if (!revision && a.suite.suite_hash !== b.suite.suite_hash) warnings.push('suite_hash differs — the eval suite changed; per-case comparison may be misleading.');
|
|
122
227
|
if (a.skill.name !== b.skill.name) warnings.push(`different skills (${a.skill.name} vs ${b.skill.name}) — comparison is not meaningful.`);
|
|
123
228
|
if ((a.run.judge || {}).samples <= 1 || (b.run.judge || {}).samples <= 1) warnings.push('one or both receipts are single-sample (no bands) — non-overlap can only be trusted when both sides are sampled.');
|
|
124
|
-
|
|
229
|
+
// WHAT SUPPRESSED THE VERDICT IS SAID, AND IT IS SAID CORRECTLY (spec 016 AC-3
|
|
230
|
+
// and AC-4, closing F-015-B).
|
|
231
|
+
//
|
|
232
|
+
// TWO DIFFERENT THINGS can suppress a verdict, and this line used to describe
|
|
233
|
+
// only one of them. A REFUSAL — the baseline-reproduction precondition — set
|
|
234
|
+
// `measured` false and then the caveat rendered `belowTested`, which on a
|
|
235
|
+
// refused pair is EMPTY: the page read `verdicts NOT computed — .` under a
|
|
236
|
+
// headline claiming `0 receipt(s) below TESTED`, on a pair where both receipts
|
|
237
|
+
// were TESTED. An empty reason and a false statement, in the artifact a report's
|
|
238
|
+
// verdicts come from. The refusal's own cause-honest reason was computed into
|
|
239
|
+
// `refusal.reason` and thrown away by the renderer, so nothing a reader could
|
|
240
|
+
// see said why the comparison had stopped.
|
|
241
|
+
//
|
|
242
|
+
// Each half is proved load-bearing by a mutation that patches THIS source in a
|
|
243
|
+
// disposable copy. An earlier revision guarded them with const-true flags and
|
|
244
|
+
// this sentence described those; the flags were removed because a
|
|
245
|
+
// defect-restoring branch resident in the published tree is a hazard, and the
|
|
246
|
+
// sentence outlived them by one commit.
|
|
247
|
+
if (refusal) {
|
|
248
|
+
warnings.push(`verdicts NOT computed — the comparison was REFUSED before any verdict was formed: ${refusal.reason}`);
|
|
249
|
+
}
|
|
250
|
+
if (belowTested.length) {
|
|
251
|
+
warnings.push(`verdicts NOT computed — ${belowTested.map(([l, r]) => `${l} is ${levelOf(r)}${r.run && r.run.source ? ` (${r.run.source})` : ''}`).join('; ')}. Drift verdicts require TESTED receipts on both sides; declared numbers are shown as context only (see /interop.html).`);
|
|
252
|
+
}
|
|
125
253
|
// Cross-provider / cross-surface disclosure (Phase 6). A comparison across
|
|
126
254
|
// providers is a skill-DURABILITY comparison across substrates, not model drift
|
|
127
255
|
// over time; across surfaces, sampling control differs. Both are flagged so a
|
|
128
256
|
// reader never mistakes one for the other (see docs/neutrality.html).
|
|
129
|
-
if ((a.run.provider || 'anthropic') !== (b.run.provider || 'anthropic')) warnings.push(`different providers (${a.run.provider || 'anthropic'} vs ${b.run.provider || 'anthropic'}) — this is a cross-substrate durability comparison, not model drift over time; read the delta, not absolute scores (see the neutrality policy).`);
|
|
130
|
-
if (a.run.surface !== b.run.surface) warnings.push(`different surfaces (${a.run.surface} vs ${b.run.surface}) — sampling control differs between surfaces; compare with care.`);
|
|
257
|
+
if (!revision && (a.run.provider || 'anthropic') !== (b.run.provider || 'anthropic')) warnings.push(`different providers (${a.run.provider || 'anthropic'} vs ${b.run.provider || 'anthropic'}) — this is a cross-substrate durability comparison, not model drift over time; read the delta, not absolute scores (see the neutrality policy).`);
|
|
258
|
+
if (!revision && a.run.surface !== b.run.surface) warnings.push(`different surfaces (${a.run.surface} vs ${b.run.surface}) — sampling control differs between surfaces; compare with care.`);
|
|
259
|
+
// The legend for the band markers, printed whenever either side carries one.
|
|
260
|
+
// A two-letter marker a reader cannot decode is worse than no marker.
|
|
261
|
+
// GATED ON WHETHER A LABEL IS ACTUALLY RENDERED, not on whether the data
|
|
262
|
+
// carries a source. A legend explains markers on the page; if `bandStr` emits
|
|
263
|
+
// none, the legend is describing something the reader cannot see. Asked of
|
|
264
|
+
// `bandStr` itself rather than recomputed, so the two cannot disagree.
|
|
265
|
+
const anyLabelRendered = [...Object.values(aB), ...Object.values(bB)]
|
|
266
|
+
.some((x) => x && /\([a-z]+\)\s*$/.test(bandStr(x)));
|
|
267
|
+
if (anyLabelRendered) {
|
|
268
|
+
warnings.push('band provenance: `(generation)` is an ACROSS-DRAW spread — receipt spec v0.5, n generation draws per arm. `(legacy)` is a JUDGE-SAMPLE spread over a single generation, which is what v0.4 and earlier recorded. They are different statistics. The comparison is the only one the older receipt admits, and it is not like for like.');
|
|
269
|
+
}
|
|
131
270
|
if (warnings.length) {
|
|
132
271
|
L.push('> **⚠ Caveats**');
|
|
133
272
|
for (const w of warnings) L.push(`> - ${w}`);
|
|
@@ -139,10 +278,23 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B' } = {}) {
|
|
|
139
278
|
L.push(`## Headline`);
|
|
140
279
|
L.push('');
|
|
141
280
|
if (!measured) {
|
|
142
|
-
|
|
281
|
+
// THE HEADLINE NAMES THE CAUSE THAT ACTUALLY APPLIES. `belowTested.length`
|
|
282
|
+
// was printed unconditionally, so a refused pair got "0 receipt(s) below
|
|
283
|
+
// TESTED" — a statement measurably false of the receipts it was given.
|
|
284
|
+
if (refusal) {
|
|
285
|
+
// The reason is a complete sentence and already ends by saying no verdict
|
|
286
|
+
// is asserted; prefixing that again produced "REFUSED — no verdict is
|
|
287
|
+
// asserted. the baseline arm did not reproduce…" — a duplicated clause and
|
|
288
|
+
// a lower-case sentence start, in the artifact a published report quotes.
|
|
289
|
+
L.push(`**REFUSED — ${refusal.reason}**`);
|
|
290
|
+
} else if (belowTested.length) {
|
|
291
|
+
L.push(`**NOT MEASURED — ${belowTested.length} receipt(s) below TESTED. Drift verdicts require TESTED receipts on both sides; the declared numbers above are context, not band-verified evidence.**`);
|
|
292
|
+
} else {
|
|
293
|
+
L.push('**NOT MEASURED — no verdict is asserted.**');
|
|
294
|
+
}
|
|
143
295
|
L.push('');
|
|
144
296
|
} else {
|
|
145
|
-
L.push(`**${headlineVerdict(perCase)}**`);
|
|
297
|
+
L.push(`**${revision ? revisionHeadline(perCase) : headlineVerdict(perCase)}**`);
|
|
146
298
|
L.push('');
|
|
147
299
|
L.push(`with_skill mean moved ${fmt(headlineDelta)} (${bandStr(aAgg)} → ${bandStr(bAgg)}; band = suite dispersion). Per-case band-overlap verdicts: ${regressions.length} regression(s), ${perCase.filter((r) => r.verdict === 'improvement').length} improvement(s), ${nWithin} within noise${nFloor ? ` (${nFloor} of them band-separated but below the ${EFFECT_FLOOR} effect floor)` : ''}.`);
|
|
148
300
|
L.push('');
|
|
@@ -167,7 +319,12 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B' } = {}) {
|
|
|
167
319
|
L.push('');
|
|
168
320
|
}
|
|
169
321
|
|
|
170
|
-
return {
|
|
322
|
+
return {
|
|
323
|
+
markdown: L.join('\n'), perCase, headlineDelta, regressions,
|
|
324
|
+
refused: !!refusal,
|
|
325
|
+
refusal_reason: refusal ? refusal.reason : null,
|
|
326
|
+
refusal_key: refusal ? refusal.key : null,
|
|
327
|
+
};
|
|
171
328
|
}
|
|
172
329
|
|
|
173
|
-
module.exports = { buildDriftReport, withSkillBands };
|
|
330
|
+
module.exports = { buildDriftReport, withSkillBands, revisionPairProblem };
|
package/lib/hygiene.js
ADDED
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
//
|
|
3
|
+
// The single definition of what counts as a leak.
|
|
4
|
+
//
|
|
5
|
+
// Two callers scan for the same classes of secret: `tests/gate.js`, which walks
|
|
6
|
+
// the working tree at gate time, and `scripts/merge-check.js`, which walks the
|
|
7
|
+
// candidate tree read out of the commit before permitting a merge. They must
|
|
8
|
+
// agree. Two copies of a deny-list drift, and the copy that quietly stops
|
|
9
|
+
// matching still reads as protection — the same argument the acknowledgment
|
|
10
|
+
// deny-list bite test already makes, applied to the scanner itself.
|
|
11
|
+
//
|
|
12
|
+
// Backlog B-8 records four approval records that reached `dev` carrying a host
|
|
13
|
+
// path. The recorded cause — "the scan walks tracked files" — is false, and this
|
|
14
|
+
// module's existence does not depend on it: `listFiles()` in the repo gate has
|
|
15
|
+
// always enumerated untracked files too, and a planted path in an untracked
|
|
16
|
+
// evidence record does take the repo gate red. The real window is VANTAGE. An
|
|
17
|
+
// approval session measures in a disposable copy made before it writes its
|
|
18
|
+
// record into the shared checkout, so the record is simply not in the tree that
|
|
19
|
+
// was scanned. Scanning the working tree harder cannot close that. Scanning the
|
|
20
|
+
// commit that is about to merge does, and that is what merge-check now uses this
|
|
21
|
+
// for.
|
|
22
|
+
//
|
|
23
|
+
// Patterns are written so a regex literal cannot match its own text: a
|
|
24
|
+
// metacharacter or a character class follows each fixed prefix.
|
|
25
|
+
|
|
26
|
+
const dec = (b64) => JSON.parse(Buffer.from(b64, 'base64').toString('utf8'));
|
|
27
|
+
|
|
28
|
+
// Build-box host + private IP, base64 so the plaintext never lands in a file.
|
|
29
|
+
const HOSTIP = dec('WyJpcC0xNzItMzEtNDAtMTU5LmFwLXNvdXRoZWFzdC0xLmNvbXB1dGUuaW50ZXJuYWwiLCIxNzIuMzEuNDAuMTU5Il0=');
|
|
30
|
+
|
|
31
|
+
const EMAIL_ALLOW = new Set(['example.com', 'example.org', 'driftproofhq.com']);
|
|
32
|
+
|
|
33
|
+
const PATTERNS = [
|
|
34
|
+
{ name: 'home-path', re: /\/home\/[a-z0-9_-]+\/|\/Users\/[A-Za-z0-9_-]+\// },
|
|
35
|
+
{ name: 'ec2-internal-host', re: /ip-\d+-\d+-\d+-\d+\.[a-z0-9.-]*compute\.(internal|amazonaws\.com)/i },
|
|
36
|
+
{ name: 'private-ip', re: /\b(10\.\d{1,3}\.\d{1,3}\.\d{1,3}|172\.(1[6-9]|2\d|3[01])\.\d{1,3}\.\d{1,3}|192\.168\.\d{1,3}\.\d{1,3})\b/ },
|
|
37
|
+
{ name: 'anthropic-key', re: /sk-ant-[A-Za-z0-9_-]{8,}/ },
|
|
38
|
+
{ name: 'github-token', re: /gh[pousr]_[A-Za-z0-9]{20,}/ },
|
|
39
|
+
{ name: 'aws-key', re: /\bAKIA[0-9A-Z]{16}\b/ },
|
|
40
|
+
{ name: 'private-key-block', re: /-----BEGIN [A-Z ]*PRIVATE KEY-----/ },
|
|
41
|
+
{ name: 'env-secret-assignment', re: /\b(ANTHROPIC_API_KEY|AWS_SECRET_ACCESS_KEY|OPENAI_API_KEY)\s*=\s*\S+/ },
|
|
42
|
+
// Codex subscription auth material (~/.codex/auth.json contents) must NEVER
|
|
43
|
+
// land in a committed file: the id_token/access_token are JWTs, and an OpenAI
|
|
44
|
+
// secret key is sk-proj-/sk-svcacct-/sk-admin-. We ban the CONTENTS (tokens),
|
|
45
|
+
// not the documented path string (`~/.codex/auth.json` is referenced in help
|
|
46
|
+
// text and docs by design).
|
|
47
|
+
{ name: 'jwt-token', re: /\beyJ[A-Za-z0-9_=-]{10,}\.eyJ[A-Za-z0-9_=-]{10,}\.[A-Za-z0-9_=-]{6,}/ },
|
|
48
|
+
{ name: 'openai-secret-key', re: /\bsk-(proj|svcacct|admin)-[A-Za-z0-9_-]{20,}/ },
|
|
49
|
+
];
|
|
50
|
+
|
|
51
|
+
// A path is a hit on its own name, with no content read: a committed .env file
|
|
52
|
+
// is a leak whatever it happens to contain.
|
|
53
|
+
function scanPath(rel) {
|
|
54
|
+
return /(^|\/)\.env(\.|$)/.test(rel) ? [{ file: rel, kind: 'env-file' }] : [];
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
function scanContent(rel, content) {
|
|
58
|
+
const hits = [];
|
|
59
|
+
for (const p of PATTERNS) {
|
|
60
|
+
const m = content.match(p.re);
|
|
61
|
+
if (m) hits.push({ file: rel, kind: p.name, sample: m[0].slice(0, 24) });
|
|
62
|
+
}
|
|
63
|
+
for (const hv of HOSTIP) if (content.includes(hv)) hits.push({ file: rel, kind: 'build-host-or-ip' });
|
|
64
|
+
const emails = content.match(/\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b/g) || [];
|
|
65
|
+
for (const e of emails) {
|
|
66
|
+
const dom = e.split('@')[1].toLowerCase();
|
|
67
|
+
if (!EMAIL_ALLOW.has(dom) && !dom.endsWith('.example')) hits.push({ file: rel, kind: 'email', sample: e });
|
|
68
|
+
}
|
|
69
|
+
return hits;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
// scanFiles(files, read, readLink) — `read(rel)` returns the file's text, or
|
|
73
|
+
// throws/returns null for anything unreadable (binary, deleted, a symlink to
|
|
74
|
+
// nowhere), which is skipped exactly as the gate has always skipped it.
|
|
75
|
+
//
|
|
76
|
+
// `readLink(rel)` is optional and returns a symbolic link's TARGET PATH as a
|
|
77
|
+
// string, or null for an entry that is not a link. Where it is supplied, that
|
|
78
|
+
// target is scanned as the entry's content through `scanContent`, so every
|
|
79
|
+
// pattern above applies to it with no second deny-list to drift (spec 011,
|
|
80
|
+
// AC-2). A link's target is a path, and a path is exactly the class of leak the
|
|
81
|
+
// `home-path` pattern exists to catch: P-1 shipped `node_modules ->
|
|
82
|
+
// /home/<user>/... ` into a public tree past a scan that read the link as an
|
|
83
|
+
// unreadable file and skipped it.
|
|
84
|
+
//
|
|
85
|
+
// The argument is optional so a caller supplying nothing behaves exactly as
|
|
86
|
+
// before. That compatibility is deliberate, and it is also the standing risk: a
|
|
87
|
+
// caller added later is blind by default. The repo gate asserts that every call
|
|
88
|
+
// site supplies a reader, with `scripts/merge-check.js` the one exception — its
|
|
89
|
+
// reader is `git show <ref>:<path>`, and git stores a link's blob as its target
|
|
90
|
+
// string, so the target already arrives as content there.
|
|
91
|
+
function scanFiles(files, read, readLink) {
|
|
92
|
+
const hits = [];
|
|
93
|
+
for (const rel of files) {
|
|
94
|
+
hits.push(...scanPath(rel));
|
|
95
|
+
// The link branch is a single condition on purpose: it is the mutation seam
|
|
96
|
+
// the spec-011 gate neutralises to prove the scanner goes blind without it.
|
|
97
|
+
if (typeof readLink === 'function') {
|
|
98
|
+
let target;
|
|
99
|
+
try { target = readLink(rel); } catch (_e) { target = null; }
|
|
100
|
+
if (typeof target === 'string') {
|
|
101
|
+
hits.push(...scanContent(rel, target));
|
|
102
|
+
continue;
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
let c;
|
|
106
|
+
try { c = read(rel); } catch (_e) { continue; }
|
|
107
|
+
if (typeof c !== 'string') continue;
|
|
108
|
+
hits.push(...scanContent(rel, c));
|
|
109
|
+
}
|
|
110
|
+
return hits;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
module.exports = { PATTERNS, EMAIL_ALLOW, HOSTIP, scanPath, scanContent, scanFiles };
|
package/lib/judge.js
CHANGED
|
@@ -99,7 +99,12 @@ async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature
|
|
|
99
99
|
// { samples:[scores], mean, stddev, reason, judge_settings, model_id, rubric_hash }
|
|
100
100
|
// `mean` ± `stddev` is the per-case confidence band (raw spread of the N scores)
|
|
101
101
|
// used by the borderline-outcome rule and per-case drift band-overlap logic.
|
|
102
|
-
|
|
102
|
+
// NO DEFAULT TIMEOUT HERE (spec 017 AC-2). This defaulted to 120000, which
|
|
103
|
+
// outranked the per-surface policy exactly as lib/run.js's literal did — so the
|
|
104
|
+
// JUDGE calls timed out on the api policy while running on a CLI surface, which
|
|
105
|
+
// is the second shadowing site and the one #007's prep session had not found.
|
|
106
|
+
// Passing `undefined` through lets lib/provider.js resolve the declared policy.
|
|
107
|
+
async function gradeSamples({ task, response, rubric, model, samples = 5, timeoutMs }) {
|
|
103
108
|
const settings = judgeSettings(samples, model);
|
|
104
109
|
const scores = [];
|
|
105
110
|
const reasons = [];
|
package/lib/receipt.js
CHANGED
|
@@ -15,7 +15,12 @@ const SCHEMA_FILES = {
|
|
|
15
15
|
'0.2': 'receipt.v0.2.schema.json',
|
|
16
16
|
'0.3': 'receipt.v0.3.schema.json',
|
|
17
17
|
'0.3.1': 'receipt.v0.3.1.schema.json',
|
|
18
|
-
|
|
18
|
+
// v0.4 moved from the unversioned filename to a version-pinned one when v0.5
|
|
19
|
+
// took the current pointer. Without this the archive would lose v0.4: every
|
|
20
|
+
// published v0.4 receipt asserts conformance by NUMBER, and the number has to
|
|
21
|
+
// keep resolving to the schema it meant (AC-10).
|
|
22
|
+
'0.4': 'receipt.v0.4.schema.json',
|
|
23
|
+
'0.5': 'receipt.schema.json',
|
|
19
24
|
};
|
|
20
25
|
|
|
21
26
|
const _validators = {};
|
|
@@ -84,15 +89,70 @@ function aggregate(caseResults) {
|
|
|
84
89
|
function buildReceipt({ skill, suite, run, cases, economics = null, verificationLevel = 'TESTED', editorialReviews = null }) {
|
|
85
90
|
// v0.3.1: cases marked failed_timeout are recorded in results.cases but EXCLUDED
|
|
86
91
|
// from aggregates — a band is never fabricated from a case that did not complete.
|
|
87
|
-
|
|
88
|
-
|
|
92
|
+
//
|
|
93
|
+
// PAIRWISE, NOT PER ARM (spec 017 AC-5, AC-6). This filtered a FLAT list and
|
|
94
|
+
// then split by mode, so a case whose baseline failed kept its with_skill arm:
|
|
95
|
+
// #007's cell 3 recorded with_skill case_count 7 against baseline 6, and its
|
|
96
|
+
// headline `delta` of +0.116 was a 7-case mean minus a 6-case mean.
|
|
97
|
+
//
|
|
98
|
+
// `comparison.delta` is PAIRED BY CONSTRUCTION — one suite, measured twice —
|
|
99
|
+
// and a paired statistic computed over unequal sets is not the statistic it
|
|
100
|
+
// names. So an arm that cannot be measured removes its CASE from both sides.
|
|
101
|
+
//
|
|
102
|
+
// Exclusion, not refusal, and the reason is recorded rather than argued: an
|
|
103
|
+
// aggregate is a summary statistic, not a verdict, so the "refuse rather than
|
|
104
|
+
// assert" rule that governs verdicts does not reach it; and refusing the whole
|
|
105
|
+
// aggregate would discard thirteen sound arms because one failed. What the
|
|
106
|
+
// aggregate owes a reader instead is that it says what it covered, which
|
|
107
|
+
// `excluded_cases` provides.
|
|
108
|
+
//
|
|
109
|
+
// The excluded case STAYS in `results.cases`. It is removed from the mean, not
|
|
110
|
+
// from the record — deleting the evidence of a failure is a different and worse
|
|
111
|
+
// defect than averaging over it.
|
|
112
|
+
const armUnusable = (c) => c.case_status === 'failed_timeout'
|
|
113
|
+
|| (c.mean == null && c.score == null);
|
|
114
|
+
const excludedIds = new Map();
|
|
115
|
+
for (const c of cases) {
|
|
116
|
+
if (!armUnusable(c)) continue;
|
|
117
|
+
if (excludedIds.has(c.id)) { excludedIds.get(c.id).modes.push(c.mode); continue; }
|
|
118
|
+
excludedIds.set(c.id, {
|
|
119
|
+
id: c.id,
|
|
120
|
+
modes: [c.mode],
|
|
121
|
+
reason: c.reason
|
|
122
|
+
|| (c.generation && c.generation.stopping_reason === 'unmeasured_exhausted'
|
|
123
|
+
? 'every generation draw was unmeasured'
|
|
124
|
+
: 'the arm has no measured result'),
|
|
125
|
+
});
|
|
126
|
+
}
|
|
127
|
+
const okCases = cases.filter((c) => !excludedIds.has(c.id));
|
|
89
128
|
const withSkill = okCases.filter((c) => c.mode === 'with_skill');
|
|
90
129
|
const baseline = okCases.filter((c) => c.mode === 'baseline');
|
|
130
|
+
const excludedCases = [...excludedIds.values()];
|
|
131
|
+
// `run.failed_case_count` COUNTS CASES, and is computed FROM the list it
|
|
132
|
+
// summarises rather than alongside it, so the two fields cannot disagree.
|
|
133
|
+
//
|
|
134
|
+
// It was `cases.length - okCases.length`, which counted ROWS. That was the
|
|
135
|
+
// same number while exclusion was per arm; once AC-5 made exclusion pairwise
|
|
136
|
+
// the subtraction removed BOTH arms of every excluded case, so one failed arm
|
|
137
|
+
// reported 2. Measured on the archive before this fix: #006's writing-plans
|
|
138
|
+
// receipt recomputed to 4 against the 2 it records, and #007's to 2 against 1
|
|
139
|
+
// — two published figures doubled by a change that never named this field.
|
|
140
|
+
//
|
|
141
|
+
// The field's name says cases and the aggregates exclude by case, so the
|
|
142
|
+
// count is the length of `results.aggregates.excluded_cases` and nothing
|
|
143
|
+
// else. A case whose BOTH arms failed is one exclusion and counts once.
|
|
144
|
+
const failedCount = excludedCases.length;
|
|
91
145
|
const aggWith = aggregate(withSkill);
|
|
92
146
|
const aggBase = aggregate(baseline);
|
|
93
147
|
|
|
94
148
|
const receipt = {
|
|
95
149
|
schema_version: RECEIPT_SCHEMA_VERSION,
|
|
150
|
+
// v0.5 CAPABILITY FLAG (F-014-F). DERIVED FROM THE CASES THEMSELVES, never
|
|
151
|
+
// from a caller's argument: the exposure this closes is a receipt asserting
|
|
152
|
+
// something it does not carry, and a flag taken on trust from the caller
|
|
153
|
+
// would be the same defect with an extra step. Absent when no case carries a
|
|
154
|
+
// draw set, which is what keeps every legacy and imported receipt valid.
|
|
155
|
+
...(cases.some((c) => c && c.generation) ? { generation_sampled: true } : {}),
|
|
96
156
|
skill: {
|
|
97
157
|
name: skill.name,
|
|
98
158
|
version: skill.version,
|
|
@@ -102,6 +162,12 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
|
|
|
102
162
|
format: suite.format,
|
|
103
163
|
suite_hash: suite.suiteHash,
|
|
104
164
|
case_count: suite.caseCount,
|
|
165
|
+
// v0.5: the per-suite canary. THIS is the canonical assembly — the runner
|
|
166
|
+
// built the field and this function dropped it, so the live smoke emitted
|
|
167
|
+
// `canary: undefined` and the schema, which makes it optional, said
|
|
168
|
+
// nothing. Caught by reading the receipt a real run produced, not by a
|
|
169
|
+
// gate; the assertion that would have caught it is added with the fix.
|
|
170
|
+
...(suite.canary ? { canary: suite.canary } : {}),
|
|
105
171
|
},
|
|
106
172
|
run: {
|
|
107
173
|
model_id: run.model_id,
|
|
@@ -120,7 +186,13 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
|
|
|
120
186
|
},
|
|
121
187
|
results: {
|
|
122
188
|
cases,
|
|
123
|
-
aggregates: {
|
|
189
|
+
aggregates: {
|
|
190
|
+
with_skill: aggWith,
|
|
191
|
+
baseline: aggBase,
|
|
192
|
+
// Present only when something was excluded, so a clean run's receipt is
|
|
193
|
+
// unchanged and the archive does not acquire an empty field.
|
|
194
|
+
...(excludedCases.length ? { excluded_cases: excludedCases } : {}),
|
|
195
|
+
},
|
|
124
196
|
},
|
|
125
197
|
comparison: {
|
|
126
198
|
with_skill_score: aggWith.mean_score,
|