driftproof 0.5.0 → 0.7.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +104 -29
- package/bin/driftproof +46 -12
- package/config.js +49 -5
- package/lib/canary.js +27 -0
- package/lib/cost.js +20 -2
- package/lib/diff.js +172 -15
- package/lib/hygiene.js +113 -0
- package/lib/judge.js +6 -1
- package/lib/receipt.js +76 -4
- package/lib/reuse.js +134 -0
- package/lib/revision.js +169 -0
- package/lib/run.js +165 -45
- package/lib/sampling.js +110 -0
- package/lib/stats.js +16 -1
- package/lib/value.js +27 -2
- package/package.json +2 -2
- package/spec/RECEIPT.md +102 -3
- package/spec/receipt.schema.json +360 -3
- package/spec/receipt.v0.4.schema.json +961 -0
package/lib/reuse.js
ADDED
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// Rerun / regrade / reuse, and the baseline-reproduction precondition.
|
|
5
|
+
// Receipt spec v0.5.
|
|
6
|
+
//
|
|
7
|
+
// TWO THINGS LIVE HERE, both decided from what two receipts RECORD rather than
|
|
8
|
+
// from a flag someone passed:
|
|
9
|
+
//
|
|
10
|
+
// 1. THE PRECONDITION (AC-6). Report #006's whole result. The no-skill arm
|
|
11
|
+
// contains no skill text, so a skill revision cannot move it. If it fails
|
|
12
|
+
// to reproduce the arm an earlier receipt measured on the same model, the
|
|
13
|
+
// same surface and the same suite, then whatever else changed, the two
|
|
14
|
+
// receipts are not measuring the same thing — and a verdict computed across
|
|
15
|
+
// them would be a comparison whose premise was never checked. That is the
|
|
16
|
+
// F-009-X class. The comparison is REFUSED and no verdict is asserted.
|
|
17
|
+
//
|
|
18
|
+
// 2. THE TRIAGE (AC-8). Terminal-Bench 4.0 distinguishes rerunning a task,
|
|
19
|
+
// regrading an existing transcript, and reusing a prior result. Adopted,
|
|
20
|
+
// translated: our unit is a (case, arm) draw set, and the trigger is the
|
|
21
|
+
// receipt's own recorded provenance — model, suite and skill content force
|
|
22
|
+
// a rerun because the generation would differ; judge and rubric force a
|
|
23
|
+
// regrade because only the scoring would; neither changing permits reuse.
|
|
24
|
+
//
|
|
25
|
+
// PURE: two receipts in, a decision out. No I/O, no provider. F-009-K's lesson
|
|
26
|
+
// applied — the decision derives from the artifact, never from a filename.
|
|
27
|
+
|
|
28
|
+
const { round } = require('./stats');
|
|
29
|
+
|
|
30
|
+
// EVERY REASON STATES WHAT WAS OBSERVED AND STOPS THERE (F-009-N). The control
|
|
31
|
+
// proves non-reproduction; it cannot say why. A reason that named a cause — "the
|
|
32
|
+
// skill regressed", "the model got worse" — would be a finding this instrument
|
|
33
|
+
// did not measure, which is the one thing it must never publish.
|
|
34
|
+
const REFUSAL_REASONS = {
|
|
35
|
+
baseline_did_not_reproduce: ({ observed, expected } = {}) =>
|
|
36
|
+
`the baseline arm did not reproduce: this run observed ${observed}, the earlier receipt recorded ${expected}, and the bands do not overlap. No verdict is asserted; the control shows non-reproduction and cannot establish a cause.`,
|
|
37
|
+
baseline_missing: () =>
|
|
38
|
+
'no baseline arm is present in one of the two receipts, so reproduction cannot be checked and no verdict is asserted.',
|
|
39
|
+
baseline_unmeasured: () =>
|
|
40
|
+
'the baseline arm has no measured draws, so reproduction cannot be checked and no verdict is asserted.',
|
|
41
|
+
incomparable_surface: () =>
|
|
42
|
+
'the two receipts record different surfaces, so the earlier measurement cannot stand as a control here and no verdict is asserted.',
|
|
43
|
+
};
|
|
44
|
+
|
|
45
|
+
const casesOf = (r) => ((r && r.results && r.results.cases) || []);
|
|
46
|
+
const armOf = (r, mode) => casesOf(r).filter((c) => c.mode === mode);
|
|
47
|
+
|
|
48
|
+
// THE ONE BAND DEFINITION, and it resolves the shape the receipt actually
|
|
49
|
+
// recorded (spec 016 AC-1, closing F-015-C).
|
|
50
|
+
//
|
|
51
|
+
// WHAT WENT WRONG. This read `c.generation.mean` and nothing else. That block
|
|
52
|
+
// exists on v0.5 receipts and on no earlier one — v0.4 records the same
|
|
53
|
+
// measurement as `c.mean` and `c.stddev` — so every v0.4 case resolved to `null`
|
|
54
|
+
// and `baselineReproduces` returned `baseline_unmeasured` for EVERY v0.4→v0.5
|
|
55
|
+
// pair, always, whatever the baselines had done. The precondition written
|
|
56
|
+
// specifically for cross-version comparison could not pass against the archive
|
|
57
|
+
// it exists to compare against, and Report #007's entire comparison step is
|
|
58
|
+
// v0.4-against-v0.5.
|
|
59
|
+
//
|
|
60
|
+
// It survived 487 repo assertions and three approval rounds because every
|
|
61
|
+
// assertion that ever exercised it handed it a fixture built to the CURRENT
|
|
62
|
+
// schema. That is the seventh narrowing class, `fixture-vs-real-artifact`.
|
|
63
|
+
//
|
|
64
|
+
// THE BAND CARRIES WHICH SHAPE IT CAME FROM, and the two are not the same
|
|
65
|
+
// statistic: v0.5's `sd` is spread ACROSS DRAWS, v0.4's `stddev` is spread across
|
|
66
|
+
// JUDGE SAMPLES of one draw. Comparing them is the only comparison v0.4 admits —
|
|
67
|
+
// it is the band that version measured — but a reader is owed the fact that the
|
|
68
|
+
// older side is a judge-level band, so `source` travels with it and the differ
|
|
69
|
+
// says so rather than presenting them as like for like.
|
|
70
|
+
//
|
|
71
|
+
// NO SWITCH GUARDS THE LEGACY PATH. An earlier revision carried a const-true
|
|
72
|
+
// `LEGACY_BAND_ENABLED` so a probe could disable the fallback without editing the
|
|
73
|
+
// logic; that leaves the pre-fix behaviour resident in the shipped tree, which an
|
|
74
|
+
// approval called what it is — a hazard a comment does not remove. The mutation
|
|
75
|
+
// probe patches this source in a disposable copy instead, and fails loudly if its
|
|
76
|
+
// anchor moves.
|
|
77
|
+
function bandOf(c) {
|
|
78
|
+
if (!c) return null;
|
|
79
|
+
const g = c.generation;
|
|
80
|
+
if (g && typeof g.mean === 'number') {
|
|
81
|
+
const sd = typeof g.sd === 'number' ? g.sd : 0;
|
|
82
|
+
return { mean: g.mean, sd, lo: g.mean - sd, hi: g.mean + sd, n: g.n_measured, source: 'generation' };
|
|
83
|
+
}
|
|
84
|
+
// v0.4 and earlier: the same measurement, under the names that version used.
|
|
85
|
+
// `score` is v0.1's spelling of the mean and is accepted for the same reason.
|
|
86
|
+
const mean = typeof c.mean === 'number' ? c.mean : (typeof c.score === 'number' ? c.score : null);
|
|
87
|
+
if (mean === null) return null;
|
|
88
|
+
const sd = typeof c.stddev === 'number' ? c.stddev : 0;
|
|
89
|
+
return { mean, sd, lo: mean - sd, hi: mean + sd, n: Array.isArray(c.samples) ? c.samples.length : null, source: 'legacy' };
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
function baselineReproduces(older, newer) {
|
|
93
|
+
const a = armOf(older, 'baseline');
|
|
94
|
+
const b = armOf(newer, 'baseline');
|
|
95
|
+
if (!a.length || !b.length) return { ok: false, key: 'baseline_missing' };
|
|
96
|
+
const bad = [];
|
|
97
|
+
for (const nb of b) {
|
|
98
|
+
const ob = a.find((x) => x.id === nb.id);
|
|
99
|
+
if (!ob) continue;
|
|
100
|
+
const bandA = bandOf(ob); const bandB = bandOf(nb);
|
|
101
|
+
if (!bandA || !bandB) return { ok: false, key: 'baseline_unmeasured' };
|
|
102
|
+
const overlap = bandA.lo <= bandB.hi && bandB.lo <= bandA.hi;
|
|
103
|
+
if (!overlap) bad.push({ id: nb.id, observed: round(bandB.mean), expected: round(bandA.mean) });
|
|
104
|
+
}
|
|
105
|
+
if (bad.length) return { ok: false, key: 'baseline_did_not_reproduce', ...bad[0], cases: bad };
|
|
106
|
+
return { ok: true };
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
// The verdict path. REFUSED is a RESULT, not an error: it carries no delta, and
|
|
110
|
+
// it is what #006 published three times.
|
|
111
|
+
function compare(older, newer) {
|
|
112
|
+
const pre = baselineReproduces(older, newer);
|
|
113
|
+
if (!pre.ok) {
|
|
114
|
+
const reason = REFUSAL_REASONS[pre.key]({ observed: pre.observed, expected: pre.expected });
|
|
115
|
+
return { verdict: 'REFUSED', reason, reason_key: pre.key, delta: null, cases: pre.cases || [] };
|
|
116
|
+
}
|
|
117
|
+
const ws = (r) => { const arm = armOf(r, 'with_skill'); const b = arm.map(bandOf).filter(Boolean); return b.length ? b.reduce((s, x) => s + x.mean, 0) / b.length : null; };
|
|
118
|
+
const a = ws(older); const b = ws(newer);
|
|
119
|
+
if (a === null || b === null) return { verdict: 'REFUSED', reason: REFUSAL_REASONS.baseline_unmeasured(), reason_key: 'baseline_unmeasured', delta: null };
|
|
120
|
+
return { verdict: 'MEASURED', delta: round(b - a), baseline_reproduced: true };
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
function triage(a, b) {
|
|
124
|
+
const g = (r, p, d) => p.split('.').reduce((o, k) => (o == null ? o : o[k]), r) ?? d;
|
|
125
|
+
const changed = (p) => g(a, p) !== g(b, p);
|
|
126
|
+
if (changed('run.model_id')) return { decision: 'rerun', reason: 'the model id differs, so the generation would differ' };
|
|
127
|
+
if (changed('suite.suite_hash')) return { decision: 'rerun', reason: 'the suite content differs, so the prompts would differ' };
|
|
128
|
+
if (changed('skill.content_hash')) return { decision: 'rerun', reason: 'the skill content differs, so the with-skill generation would differ' };
|
|
129
|
+
if (changed('run.judge.model_id')) return { decision: 'regrade', reason: 'only the judge differs, so the existing generations can be rescored' };
|
|
130
|
+
if (changed('run.rubric_hash')) return { decision: 'regrade', reason: 'only the rubric differs, so the existing generations can be rescored' };
|
|
131
|
+
return { decision: 'reuse', reason: 'model, suite, skill, judge and rubric all match, so the earlier result stands' };
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
module.exports = { compare, baselineReproduces, triage, REFUSAL_REASONS, bandOf };
|
package/lib/revision.js
ADDED
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
const { bandVerdict, round } = require('./stats');
|
|
5
|
+
const { bandOf } = require('./reuse');
|
|
6
|
+
const { EFFECT_FLOOR } = require('../config');
|
|
7
|
+
|
|
8
|
+
// Revision drift — the fifth report type (spec 009, Report #006).
|
|
9
|
+
//
|
|
10
|
+
// Every other report type holds the skill text fixed and moves something
|
|
11
|
+
// underneath it: the model release (#001, #003), the vendor surface (#002), the
|
|
12
|
+
// capability tier (#004), the axes and the price (#005). This one inverts the
|
|
13
|
+
// design. The substrate is held still — same model, same provider, same surface,
|
|
14
|
+
// same suite, same fixed judge, same sampling — and the SKILL'S OWN TEXT moves,
|
|
15
|
+
// from the revision a report pinned to the revision upstream ships today.
|
|
16
|
+
//
|
|
17
|
+
// This module holds the report type's LANGUAGE as code rather than as
|
|
18
|
+
// hand-written page copy. That is deliberate: the fairness rule below is the one
|
|
19
|
+
// a measurement project is most tempted to apply in one direction only, and a
|
|
20
|
+
// rule that lives in prose cannot be gated before the run that would tempt it.
|
|
21
|
+
|
|
22
|
+
// ── the cell headline ────────────────────────────────────────────────────────
|
|
23
|
+
// A summary of the per-case band-overlap verdicts, worded about the REVISION.
|
|
24
|
+
// The release-drift headline says "the skill is measurably weaker", which is a
|
|
25
|
+
// sentence about a skill under a moving model. Here the model is the control.
|
|
26
|
+
function revisionHeadline(perCase) {
|
|
27
|
+
const reg = perCase.filter((r) => r.verdict === 'regression').length;
|
|
28
|
+
const imp = perCase.filter((r) => r.verdict === 'improvement').length;
|
|
29
|
+
const s = (n) => (n === 1 ? '' : 's');
|
|
30
|
+
if (reg && imp) {
|
|
31
|
+
return `MIXED — the revision improved ${imp} case${s(imp)} and regressed ${reg} on non-overlapping bands.`;
|
|
32
|
+
}
|
|
33
|
+
if (reg) {
|
|
34
|
+
return `REVISION REGRESSED — ${reg} case${s(reg)} scored lower under the current upstream text (bands do not overlap).`;
|
|
35
|
+
}
|
|
36
|
+
if (imp) {
|
|
37
|
+
return `REVISION IMPROVED — ${imp} case${s(imp)} scored higher under the current upstream text (bands do not overlap); none regressed.`;
|
|
38
|
+
}
|
|
39
|
+
return 'WITHIN NOISE — the revision moved no case beyond its confidence band; the pinned text and the current text measure the same.';
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
// Classification word for a cell, from its headline. Kept separate so a caller
|
|
43
|
+
// can branch on the class without parsing prose.
|
|
44
|
+
function revisionClass(perCase) {
|
|
45
|
+
const h = revisionHeadline(perCase);
|
|
46
|
+
return h.split(' —')[0];
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
// ── the fairness sentence ────────────────────────────────────────────────────
|
|
50
|
+
// Spec 009 § Fairness, clauses 1 and 4. Where a revision IMPROVED a skill, the
|
|
51
|
+
// published #005 figure understates the pack a reader can install today; where it
|
|
52
|
+
// REGRESSED one, #005 overstates it. Both sentences are generated by the same
|
|
53
|
+
// function, from the same template, so the disclosure cannot quietly become a
|
|
54
|
+
// one-directional courtesy — the symmetry is a property of the code, and the
|
|
55
|
+
// gate asserts it.
|
|
56
|
+
//
|
|
57
|
+
// A cell within noise gets NO sentence. #005's figure stands unamended, because
|
|
58
|
+
// nothing was measured that would amend it, and manufacturing a hedge for a null
|
|
59
|
+
// result is how a report launders noise into a finding.
|
|
60
|
+
function fairnessSentence({ slug, classification, report005Delta, measuredDelta }) {
|
|
61
|
+
const cls = String(classification || '');
|
|
62
|
+
if (cls !== 'REVISION IMPROVED' && cls !== 'REVISION REGRESSED') return null;
|
|
63
|
+
const improved = cls === 'REVISION IMPROVED';
|
|
64
|
+
const direction = improved ? 'understates' : 'overstates';
|
|
65
|
+
const d = (n) => (n == null ? 'n/a' : (n >= 0 ? '+' : '') + Number(n).toFixed(3));
|
|
66
|
+
return `Report #005 measured ${slug} at ${d(report005Delta)} on the text it had pinned. `
|
|
67
|
+
+ `Report #006 measures the current upstream revision at ${d(measuredDelta)} on the same substrate and the same suite. `
|
|
68
|
+
+ `#005's published figure therefore ${direction} the pack upstream ships today for this skill, `
|
|
69
|
+
+ `and is amended by this report rather than corrected in place.`;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
// ── per-cell scoping disclosures ─────────────────────────────────────────────
|
|
73
|
+
// Spec 009 AC-10. One cell in Report #006 measures something other than what its
|
|
74
|
+
// upstream author changed it to do, and the reader looking at that row is the
|
|
75
|
+
// reader who needs to be told.
|
|
76
|
+
const SCOPING_NOTES = {
|
|
77
|
+
'git-workflow-and-versioning':
|
|
78
|
+
'This revision changes the frontmatter `description:` line. In a skill runtime a description is a '
|
|
79
|
+
+ 'routing trigger: it decides whether the skill loads, and never reaches the model as guidance. '
|
|
80
|
+
+ 'Driftproof makes no routing decision — it always injects the skill, and passes the whole file, '
|
|
81
|
+
+ 'frontmatter included, as the system prompt. This cell therefore measures the revision as added '
|
|
82
|
+
+ 'context and cannot measure it as a trigger.',
|
|
83
|
+
};
|
|
84
|
+
function scopingNote(slug) {
|
|
85
|
+
return SCOPING_NOTES[slug] || null;
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
// ── the baseline-reproduction control ────────────────────────────────────────
|
|
89
|
+
// Spec 009 AC-6, and the thing that makes the free pinned arm honest.
|
|
90
|
+
//
|
|
91
|
+
// Report #006 reuses #005's receipts as the pinned-text arm. That is valid only
|
|
92
|
+
// if the substrate has not moved, and `run.model_release_date` is null on every
|
|
93
|
+
// #005 receipt, so id equality is the only version evidence a receipt carries. A
|
|
94
|
+
// provider that re-points a concrete id at a new snapshot is invisible to it.
|
|
95
|
+
//
|
|
96
|
+
// It does not have to be. Every fresh run emits a BASELINE arm: the same cases,
|
|
97
|
+
// the same substrate, and no skill text at all. The revision cannot touch it by
|
|
98
|
+
// construction, so comparing the fresh baseline against the reused receipt's
|
|
99
|
+
// baseline re-measures exactly the thing id equality could not prove — at no
|
|
100
|
+
// extra cost, because that arm is already paid for.
|
|
101
|
+
//
|
|
102
|
+
// A cell whose baselines do not reproduce is NOT MEASURED. The reuse is a tested
|
|
103
|
+
// prediction, not an assumption the report asks the reader to grant.
|
|
104
|
+
function baselineBands(receipt) {
|
|
105
|
+
const out = {};
|
|
106
|
+
for (const c of receipt.results.cases) {
|
|
107
|
+
if (c.mode !== 'baseline') continue;
|
|
108
|
+
// ONE BAND DEFINITION (spec 016 AC-1). This built its own from the v0.4-shaped
|
|
109
|
+
// fields, which worked — and that is the point: three copies existed, two
|
|
110
|
+
// reading the legacy shape and one reading only v0.5, and the one that read
|
|
111
|
+
// only v0.5 was the one a cross-version control depended on (F-015-C). Routing
|
|
112
|
+
// every comparison path through `bandOf` means a future shape is added once.
|
|
113
|
+
const b = bandOf(c);
|
|
114
|
+
if (b) out[c.id] = { mean: b.mean, stddev: b.sd, source: b.source };
|
|
115
|
+
}
|
|
116
|
+
return out;
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
function baselineControl(reused, fresh) {
|
|
120
|
+
const A = baselineBands(reused);
|
|
121
|
+
const B = baselineBands(fresh);
|
|
122
|
+
const ids = [...new Set([...Object.keys(A), ...Object.keys(B)])];
|
|
123
|
+
|
|
124
|
+
const perCase = ids.map((id) => {
|
|
125
|
+
const before = A[id] || null;
|
|
126
|
+
const after = B[id] || null;
|
|
127
|
+
if (!before || !after) return { id, before, after, delta: null, moved: false, missing: true };
|
|
128
|
+
const delta = round(after.mean - before.mean);
|
|
129
|
+
// The same rule the study uses everywhere else: band separation AND the
|
|
130
|
+
// effect floor. A baseline that wobbles inside its band has not moved.
|
|
131
|
+
const raw = bandVerdict(before.mean, before.stddev, after.mean, after.stddev);
|
|
132
|
+
const separated = raw === 'regression' || raw === 'improvement';
|
|
133
|
+
return { id, before, after, delta, moved: separated && Math.abs(delta) >= EFFECT_FLOOR, missing: false };
|
|
134
|
+
});
|
|
135
|
+
|
|
136
|
+
const movedCases = perCase.filter((r) => r.moved);
|
|
137
|
+
const missing = perCase.filter((r) => r.missing);
|
|
138
|
+
const reproduced = movedCases.length === 0 && missing.length === 0;
|
|
139
|
+
const aggDelta = round(
|
|
140
|
+
(fresh.comparison && fresh.comparison.baseline_score != null ? fresh.comparison.baseline_score : 0)
|
|
141
|
+
- (reused.comparison && reused.comparison.baseline_score != null ? reused.comparison.baseline_score : 0),
|
|
142
|
+
);
|
|
143
|
+
|
|
144
|
+
return {
|
|
145
|
+
reproduced,
|
|
146
|
+
blocked: !reproduced,
|
|
147
|
+
verdict: reproduced ? 'MEASURED' : 'NOT MEASURED',
|
|
148
|
+
moved_cases: movedCases.map((r) => r.id),
|
|
149
|
+
missing_cases: missing.map((r) => r.id),
|
|
150
|
+
aggregate_baseline_delta: aggDelta,
|
|
151
|
+
floor: EFFECT_FLOOR,
|
|
152
|
+
// THE CONTROL PROVES NON-REPRODUCTION. IT CANNOT SAY WHY. These strings used
|
|
153
|
+
// to read 'the substrate moved' and 'the substrate held still' — a cause,
|
|
154
|
+
// asserted by a comparison that measures two baseline arms and nothing else.
|
|
155
|
+
// A 120-call stability probe then found generation-level sampling noise large
|
|
156
|
+
// enough to account for every gap this control saw, with no substrate movement
|
|
157
|
+
// required, and the report page retracted the claim while three committed
|
|
158
|
+
// control records still carried it (approval finding F-009-N). Reason strings
|
|
159
|
+
// only: no score, sample, hash or verdict changed with this edit.
|
|
160
|
+
reason: reproduced
|
|
161
|
+
? 'the fresh baseline reproduces the reused receipt\'s baseline within the band and the floor, so the pinned-arm reuse stands for this cell'
|
|
162
|
+
: `the fresh baseline does not reproduce the reused receipt's baseline (${movedCases.length} case(s) moved beyond the band and the ${EFFECT_FLOOR} floor${missing.length ? `, ${missing.length} case(s) absent on one side` : ''}) — the reused pinned arm is not comparable to the fresh arm, so revision drift cannot be separated from whatever else changed in this cell; the control establishes non-reproduction and does not identify a cause`,
|
|
163
|
+
perCase,
|
|
164
|
+
};
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
module.exports = {
|
|
168
|
+
revisionHeadline, revisionClass, fairnessSentence, scopingNote, baselineControl,
|
|
169
|
+
};
|
package/lib/run.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// SPDX-License-Identifier: Apache-2.0
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
-
const { complete, resolveModel, surfaceForModel, isMeteredSurface, CODEX_OVERHEAD_NOTE } = require('./provider');
|
|
4
|
+
const { complete, resolveModel, surfaceForModel, isMeteredSurface, retryPolicyForSurface, CODEX_OVERHEAD_NOTE } = require('./provider');
|
|
5
5
|
const { gradeSamples, judgeSettings } = require('./judge');
|
|
6
6
|
const { buildReceipt } = require('./receipt');
|
|
7
7
|
const { sha256 } = require('./canonical');
|
|
@@ -11,7 +11,9 @@ const { runChecks } = require('./checks');
|
|
|
11
11
|
const { estimateTokens } = require('./skillCost');
|
|
12
12
|
const { hasUsage, normalizeUsage } = require('./usage');
|
|
13
13
|
const { buildPricingSnapshot, computeEconomics } = require('./value');
|
|
14
|
-
const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES } = require('../config');
|
|
14
|
+
const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES, DEV_MAX_CALLS } = require('../config');
|
|
15
|
+
const { SAMPLING, acrossDraws, nextAction } = require('./sampling');
|
|
16
|
+
const { suiteCanary } = require('./canary');
|
|
15
17
|
|
|
16
18
|
// Known model release dates (best-effort; null when unknown). Recorded into the
|
|
17
19
|
// receipt so drift reports can order runs by model age. Dateless model ids
|
|
@@ -36,9 +38,14 @@ function releaseDateFor(modelId) {
|
|
|
36
38
|
}
|
|
37
39
|
|
|
38
40
|
// Calls for one model run: each case does 2 generations (with_skill + baseline)
|
|
39
|
-
// and 2×samples judge calls (each generation judged `samples` times)
|
|
40
|
-
|
|
41
|
-
|
|
41
|
+
// and 2×samples judge calls (each generation judged `samples` times) — times the
|
|
42
|
+
// number of GENERATION DRAWS per arm (v0.5).
|
|
43
|
+
//
|
|
44
|
+
// `draws` DEFAULTS TO 1 so every existing caller projects exactly what it
|
|
45
|
+
// projected before; the runner's own cost guard passes the sampling MAXIMUM,
|
|
46
|
+
// which is the fail-safe direction for a guard that decides whether to spend.
|
|
47
|
+
function projectCalls(caseCount, samples, draws = 1) {
|
|
48
|
+
return caseCount * draws * (2 + 2 * samples);
|
|
42
49
|
}
|
|
43
50
|
|
|
44
51
|
// Ask the target model to perform one eval case. `withSkill` decides whether the
|
|
@@ -126,14 +133,45 @@ async function mapPool(items, concurrency, fn) {
|
|
|
126
133
|
// keepTranscripts — when true, the run records transcripts:"retained-local"
|
|
127
134
|
// and returns the raw generations + judge outputs so the
|
|
128
135
|
// caller can write them to transcripts/<receipt-id>/.
|
|
136
|
+
// THE ONE PLACE A CALL TIMEOUT IS DECIDED (spec 017 AC-1, AC-2).
|
|
137
|
+
//
|
|
138
|
+
// PURE: a surface name and the run's options in, milliseconds out. No I/O, no
|
|
139
|
+
// clock, no provider — so a probe can ask what a run WOULD use without making a
|
|
140
|
+
// call, which is what makes this criterion assertable at all.
|
|
141
|
+
//
|
|
142
|
+
// An operator-supplied `opts.timeoutMs` still wins; what is gone is the silent
|
|
143
|
+
// literal that used to stand in for a policy. `Number.isFinite` rather than a
|
|
144
|
+
// truthiness test, so an explicit 0 is a value and not a fall-through.
|
|
145
|
+
// Declared as a NAMED FUNCTION EXPRESSION bound to a const, not as a bare
|
|
146
|
+
// declaration. The binding the module exports and the name inside the function
|
|
147
|
+
// are then separable, so a mutation probe can rename the inner function to prove
|
|
148
|
+
// the export is load-bearing without the module failing to load on an undefined
|
|
149
|
+
// identifier. The inner name is kept for stack traces.
|
|
150
|
+
const resolveCallTimeoutMs = function resolveCallTimeoutMs(surface, opts = {}) {
|
|
151
|
+
if (opts && Number.isFinite(opts.timeoutMs)) return opts.timeoutMs;
|
|
152
|
+
return retryPolicyForSurface(surface).timeoutMs;
|
|
153
|
+
};
|
|
154
|
+
|
|
129
155
|
async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
130
156
|
const modelId = resolveModel(model);
|
|
131
157
|
const judgeModel = opts.judgeModel ? resolveModel(opts.judgeModel) : modelId;
|
|
132
|
-
|
|
158
|
+
// THE SURFACE'S OWN DECLARED TIMEOUT, resolved per arm (spec 017 AC-1).
|
|
159
|
+
//
|
|
160
|
+
// This read `opts.timeoutMs || 120000`, and `lib/provider.js` documents that an
|
|
161
|
+
// explicit caller value wins — so the literal outranked the policy and the
|
|
162
|
+
// 300 s claude-cli timeout, written for cold-start-dominated CLI subprocesses,
|
|
163
|
+
// never executed. Report #007 lost 25 of 160 draws to
|
|
164
|
+
// `provider(claude-cli) timed out after 120000ms`, and one arm entirely.
|
|
165
|
+
//
|
|
166
|
+
// RESOLVED SEPARATELY FOR GENERATION AND JUDGE, because they can be different
|
|
167
|
+
// models on different surfaces: #007 generated on claude-fable-5 and judged on
|
|
168
|
+
// claude-haiku-4-5. One shared timeout would apply one surface's policy to both.
|
|
169
|
+
const genTimeoutMs = resolveCallTimeoutMs(surfaceForModel(modelId), opts);
|
|
170
|
+
const judgeTimeoutMs = resolveCallTimeoutMs(surfaceForModel(judgeModel), opts);
|
|
133
171
|
// Optional per-case timeout overrides { caseId: ms }; a slow case can get a
|
|
134
172
|
// longer budget without lengthening every other case's per-call timeout.
|
|
135
173
|
const caseTimeoutMs = opts.caseTimeoutMs || {};
|
|
136
|
-
const maxCalls = opts.maxCalls ||
|
|
174
|
+
const maxCalls = opts.maxCalls || DEV_MAX_CALLS;
|
|
137
175
|
const samples = opts.samples || DEFAULT_JUDGE_SAMPLES;
|
|
138
176
|
const concurrency = Math.max(1, opts.concurrency || 1);
|
|
139
177
|
const onProgress = opts.onProgress || (() => {});
|
|
@@ -145,7 +183,10 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
145
183
|
|
|
146
184
|
// Cost guard: project the whole run up front and refuse before spending a
|
|
147
185
|
// single call if it would blow the cap.
|
|
148
|
-
|
|
186
|
+
// The WORST CASE, deliberately: escalation is adaptive and a guard that
|
|
187
|
+
// projects the floor would wave through a run that then draws ten times.
|
|
188
|
+
// Refusing a run that would have fit is recoverable; overspending is not.
|
|
189
|
+
const projected = projectCalls(cases.length, samples, SAMPLING.max);
|
|
149
190
|
if (projected > maxCalls) {
|
|
150
191
|
const e = new Error(`cost guard: projected ${projected} calls exceeds cap ${maxCalls} (${cases.length} cases × (2 + 2×${samples} samples)). Raise --max-calls or lower --max-cases/--samples.`);
|
|
151
192
|
e.code = 'CALL_CAP';
|
|
@@ -163,43 +204,111 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
163
204
|
const genKind = (ws) => (ws ? 'gen_with_skill' : 'gen_baseline');
|
|
164
205
|
const pairs = await mapPool(tasks, concurrency, async ({ c, withSkill }) => {
|
|
165
206
|
const mode = withSkill ? 'with_skill' : 'baseline';
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
207
|
+
// A per-case override is an operator's explicit input and still wins, for
|
|
208
|
+
// both arms. Otherwise each arm uses its own surface's declared policy.
|
|
209
|
+
const ctGen = caseTimeoutMs[c.id] || genTimeoutMs;
|
|
210
|
+
const ctJudge = caseTimeoutMs[c.id] || judgeTimeoutMs;
|
|
211
|
+
// v0.5 — DRAW THE GENERATION n TIMES, and keep each draw's judge samples
|
|
212
|
+
// INSIDE that draw. Pooling k×n scores into one list is precisely what made
|
|
213
|
+
// generation noise read as judge noise: it is the defect Report #006 exists
|
|
214
|
+
// to name, and the nesting is the whole measurement.
|
|
215
|
+
const draws = [];
|
|
216
|
+
let last = null; // last MEASURED draw — carries the v0.4-shaped fields
|
|
217
|
+
let lastTranscript = null;
|
|
218
|
+
let action = { stop: false, reason: 'below_min' };
|
|
219
|
+
let fatal = null;
|
|
220
|
+
|
|
221
|
+
while (!action.stop && draws.length < SAMPLING.max) {
|
|
222
|
+
const drawIndex = draws.length;
|
|
223
|
+
try {
|
|
224
|
+
onProgress({ case: c.id, mode, phase: 'generate', draw: drawIndex });
|
|
225
|
+
const gen = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs: ctGen });
|
|
226
|
+
calls += 1;
|
|
227
|
+
if (budget) budget.add((gen.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
|
|
228
|
+
const generationHash = sha256(String(gen.text || ''));
|
|
229
|
+
onProgress({ case: c.id, mode, phase: 'judge', samples, draw: drawIndex });
|
|
230
|
+
const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ctJudge, samples });
|
|
231
|
+
calls += samples;
|
|
232
|
+
if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judgeModel, 'judge'));
|
|
233
|
+
const draw = {
|
|
234
|
+
draw_index: drawIndex,
|
|
235
|
+
generation_hash: generationHash,
|
|
236
|
+
status: 'measured',
|
|
237
|
+
samples: jr.caseResult.samples,
|
|
238
|
+
judge_sample_hashes: jr.caseResult.judge_sample_hashes,
|
|
239
|
+
mean: jr.caseResult.mean,
|
|
240
|
+
stddev: jr.caseResult.stddev,
|
|
241
|
+
};
|
|
242
|
+
if (hasUsage(gen.usage)) draw.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
|
|
243
|
+
if (jr.caseResult.judge_usage) draw.judge_usage = jr.caseResult.judge_usage;
|
|
244
|
+
draws.push(draw);
|
|
245
|
+
last = jr;
|
|
246
|
+
if (keepTranscripts) lastTranscript = { id: c.id, mode, generation: String(gen.text || ''), judge_outputs: jr.sampleTexts };
|
|
247
|
+
} catch (e) {
|
|
248
|
+
if (e && e.code === 'BUDGET_HARDSTOP') throw e; // budget hard-stop stays fatal
|
|
249
|
+
if (!isTimeout(e)) { fatal = e; break; } // non-timeout errors stay fatal
|
|
250
|
+
if (budget) {
|
|
251
|
+
try {
|
|
252
|
+
if (e.phase === 'judge') budget.add((e.judgeAttempts || 1) * perCallCostUSD(judgeModel, 'judge'));
|
|
253
|
+
else budget.add((e.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
|
|
254
|
+
} catch (be) { if (be && be.code === 'BUDGET_HARDSTOP') throw be; }
|
|
255
|
+
}
|
|
256
|
+
// F-009-L: the draw is UNMEASURED. No score, no fabricated samples, and
|
|
257
|
+
// it is excluded from every statistic rather than counted as a zero — a
|
|
258
|
+
// zero asserts a measurement, and a timeout is the absence of one.
|
|
259
|
+
draws.push({
|
|
260
|
+
draw_index: drawIndex,
|
|
261
|
+
generation_hash: null,
|
|
262
|
+
status: 'unmeasured',
|
|
263
|
+
reason: String((e && e.message) || 'timeout').slice(0, 200),
|
|
264
|
+
samples: [],
|
|
265
|
+
mean: null,
|
|
266
|
+
stddev: null,
|
|
267
|
+
});
|
|
268
|
+
onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: String((e && e.message) || 'timeout') });
|
|
198
269
|
}
|
|
270
|
+
action = nextAction(draws);
|
|
271
|
+
}
|
|
272
|
+
if (fatal) throw fatal;
|
|
273
|
+
|
|
274
|
+
const agg = acrossDraws(draws);
|
|
275
|
+
const generation = {
|
|
276
|
+
n_planned: SAMPLING.min,
|
|
277
|
+
n_drawn: agg.n_drawn,
|
|
278
|
+
n_measured: agg.n_measured,
|
|
279
|
+
n_unmeasured: agg.n_unmeasured,
|
|
280
|
+
stopping_reason: action.reason,
|
|
281
|
+
mean: agg.mean,
|
|
282
|
+
sd: agg.sd,
|
|
283
|
+
judge_sd_mean: agg.judge_sd_mean,
|
|
284
|
+
variance_ratio: agg.variance_ratio,
|
|
285
|
+
// WHICH null, when it is null (F-014-C). Copied through explicitly rather
|
|
286
|
+
// than spread from `agg`: this assembly names its keys one by one, and the
|
|
287
|
+
// canary was dropped by exactly such an assembly silently gaining a field
|
|
288
|
+
// upstream that nothing here carried down (F-014-D).
|
|
289
|
+
variance_ratio_unavailable: agg.variance_ratio_unavailable,
|
|
290
|
+
draws,
|
|
291
|
+
};
|
|
292
|
+
|
|
293
|
+
// Every draw failed: the case is recorded failed_timeout as before, but it
|
|
294
|
+
// now carries the draw list showing WHAT failed and how often.
|
|
295
|
+
if (!last) {
|
|
199
296
|
failedCases += 1;
|
|
200
|
-
|
|
201
|
-
return { caseResult: { id: c.id, mode, case_status: 'failed_timeout', reason: String((e && e.message) || 'timeout').slice(0, 200) }, transcript: null };
|
|
297
|
+
return { caseResult: { id: c.id, mode, case_status: 'failed_timeout', reason: (draws[draws.length - 1] || {}).reason || 'timeout', generation }, transcript: null };
|
|
202
298
|
}
|
|
299
|
+
|
|
300
|
+
// The v0.4-shaped fields now describe the DRAW SET, not one arbitrary draw,
|
|
301
|
+
// so a v0.4 reader pointed at a v0.5 receipt reads the aggregate rather than
|
|
302
|
+
// whichever draw happened to be last. `generation_hash`, `samples` and
|
|
303
|
+
// `judge_sample_hashes` continue to describe the last measured draw, which
|
|
304
|
+
// is the one they have always described; RECEIPT.md states this.
|
|
305
|
+
const caseResult = { ...last.caseResult, generation };
|
|
306
|
+
caseResult.mean = agg.mean;
|
|
307
|
+
caseResult.score = agg.mean;
|
|
308
|
+
caseResult.stddev = agg.sd;
|
|
309
|
+
caseResult.outcome = outcomeFor(agg.mean, agg.sd, c.pass_threshold);
|
|
310
|
+
onProgress({ case: c.id, mode, phase: 'done', outcome: caseResult.outcome, score: caseResult.mean, stddev: caseResult.stddev });
|
|
311
|
+
return { caseResult, transcript: lastTranscript };
|
|
203
312
|
});
|
|
204
313
|
const caseResults = pairs.map((p) => p.caseResult);
|
|
205
314
|
const transcripts = keepTranscripts ? pairs.map((p) => p.transcript) : null;
|
|
@@ -229,7 +338,18 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
229
338
|
// v0.3.1 value-per-token axis: estimated SKILL.md token size.
|
|
230
339
|
tokens: estimateTokens(skill.skillMd),
|
|
231
340
|
},
|
|
232
|
-
|
|
341
|
+
// v0.5: the suite canary. Derived from the suite identity and its case ids,
|
|
342
|
+
// so it is stable without a registry and distinct across suites — a leaked
|
|
343
|
+
// suite is detectable in a corpus. A detection aid, not a control.
|
|
344
|
+
suite: {
|
|
345
|
+
format: skill.suite.format,
|
|
346
|
+
suiteHash: skill.suite.suiteHash,
|
|
347
|
+
caseCount: skill.suite.caseCount,
|
|
348
|
+
// The FULL suite, never the post---max-cases list: a canary derived from a
|
|
349
|
+
// truncated run is not stable for the suite, and a canary that moves
|
|
350
|
+
// cannot say which suite leaked. Approval finding, spec 014.
|
|
351
|
+
canary: suiteCanary({ id: skill.name || skill.suite.suiteHash, cases: (skill.suite.cases || cases).map((c) => ({ id: c.id })) }),
|
|
352
|
+
},
|
|
233
353
|
run: {
|
|
234
354
|
model_id: modelId,
|
|
235
355
|
model_release_date: releaseDateFor(modelId),
|
|
@@ -277,7 +397,7 @@ function summarizeReceipt(receipt) {
|
|
|
277
397
|
const aggs = receipt.results.aggregates;
|
|
278
398
|
const sign = cmp.delta >= 0 ? '+' : '';
|
|
279
399
|
if (receipt.run.status === 'incomplete') {
|
|
280
|
-
L.push(`> ⏱ **INCOMPLETE** — ${receipt.run.failed_case_count} case(s)
|
|
400
|
+
L.push(`> ⏱ **INCOMPLETE** — ${receipt.run.failed_case_count} case(s) had an arm that could not be measured and are EXCLUDED from the aggregates below, BOTH arms together; this receipt must not be used to compute a drift/durability verdict.`);
|
|
281
401
|
L.push('');
|
|
282
402
|
}
|
|
283
403
|
L.push(`with_skill **${band(aggs.with_skill.mean_score, aggs.with_skill.stddev)}** vs baseline **${band(aggs.baseline.mean_score, aggs.baseline.stddev)}**`);
|
|
@@ -300,4 +420,4 @@ function summarizeReceipt(receipt) {
|
|
|
300
420
|
return L.join('\n');
|
|
301
421
|
}
|
|
302
422
|
|
|
303
|
-
module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor };
|
|
423
|
+
module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor, resolveCallTimeoutMs };
|