driftproof 0.5.0 → 0.7.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/reuse.js ADDED
@@ -0,0 +1,134 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // Rerun / regrade / reuse, and the baseline-reproduction precondition.
5
+ // Receipt spec v0.5.
6
+ //
7
+ // TWO THINGS LIVE HERE, both decided from what two receipts RECORD rather than
8
+ // from a flag someone passed:
9
+ //
10
+ // 1. THE PRECONDITION (AC-6). Report #006's whole result. The no-skill arm
11
+ // contains no skill text, so a skill revision cannot move it. If it fails
12
+ // to reproduce the arm an earlier receipt measured on the same model, the
13
+ // same surface and the same suite, then whatever else changed, the two
14
+ // receipts are not measuring the same thing — and a verdict computed across
15
+ // them would be a comparison whose premise was never checked. That is the
16
+ // F-009-X class. The comparison is REFUSED and no verdict is asserted.
17
+ //
18
+ // 2. THE TRIAGE (AC-8). Terminal-Bench 4.0 distinguishes rerunning a task,
19
+ // regrading an existing transcript, and reusing a prior result. Adopted,
20
+ // translated: our unit is a (case, arm) draw set, and the trigger is the
21
+ // receipt's own recorded provenance — model, suite and skill content force
22
+ // a rerun because the generation would differ; judge and rubric force a
23
+ // regrade because only the scoring would; neither changing permits reuse.
24
+ //
25
+ // PURE: two receipts in, a decision out. No I/O, no provider. F-009-K's lesson
26
+ // applied — the decision derives from the artifact, never from a filename.
27
+
28
+ const { round } = require('./stats');
29
+
30
+ // EVERY REASON STATES WHAT WAS OBSERVED AND STOPS THERE (F-009-N). The control
31
+ // proves non-reproduction; it cannot say why. A reason that named a cause — "the
32
+ // skill regressed", "the model got worse" — would be a finding this instrument
33
+ // did not measure, which is the one thing it must never publish.
34
+ const REFUSAL_REASONS = {
35
+ baseline_did_not_reproduce: ({ observed, expected } = {}) =>
36
+ `the baseline arm did not reproduce: this run observed ${observed}, the earlier receipt recorded ${expected}, and the bands do not overlap. No verdict is asserted; the control shows non-reproduction and cannot establish a cause.`,
37
+ baseline_missing: () =>
38
+ 'no baseline arm is present in one of the two receipts, so reproduction cannot be checked and no verdict is asserted.',
39
+ baseline_unmeasured: () =>
40
+ 'the baseline arm has no measured draws, so reproduction cannot be checked and no verdict is asserted.',
41
+ incomparable_surface: () =>
42
+ 'the two receipts record different surfaces, so the earlier measurement cannot stand as a control here and no verdict is asserted.',
43
+ };
44
+
45
+ const casesOf = (r) => ((r && r.results && r.results.cases) || []);
46
+ const armOf = (r, mode) => casesOf(r).filter((c) => c.mode === mode);
47
+
48
+ // THE ONE BAND DEFINITION, and it resolves the shape the receipt actually
49
+ // recorded (spec 016 AC-1, closing F-015-C).
50
+ //
51
+ // WHAT WENT WRONG. This read `c.generation.mean` and nothing else. That block
52
+ // exists on v0.5 receipts and on no earlier one — v0.4 records the same
53
+ // measurement as `c.mean` and `c.stddev` — so every v0.4 case resolved to `null`
54
+ // and `baselineReproduces` returned `baseline_unmeasured` for EVERY v0.4→v0.5
55
+ // pair, always, whatever the baselines had done. The precondition written
56
+ // specifically for cross-version comparison could not pass against the archive
57
+ // it exists to compare against, and Report #007's entire comparison step is
58
+ // v0.4-against-v0.5.
59
+ //
60
+ // It survived 487 repo assertions and three approval rounds because every
61
+ // assertion that ever exercised it handed it a fixture built to the CURRENT
62
+ // schema. That is the seventh narrowing class, `fixture-vs-real-artifact`.
63
+ //
64
+ // THE BAND CARRIES WHICH SHAPE IT CAME FROM, and the two are not the same
65
+ // statistic: v0.5's `sd` is spread ACROSS DRAWS, v0.4's `stddev` is spread across
66
+ // JUDGE SAMPLES of one draw. Comparing them is the only comparison v0.4 admits —
67
+ // it is the band that version measured — but a reader is owed the fact that the
68
+ // older side is a judge-level band, so `source` travels with it and the differ
69
+ // says so rather than presenting them as like for like.
70
+ //
71
+ // NO SWITCH GUARDS THE LEGACY PATH. An earlier revision carried a const-true
72
+ // `LEGACY_BAND_ENABLED` so a probe could disable the fallback without editing the
73
+ // logic; that leaves the pre-fix behaviour resident in the shipped tree, which an
74
+ // approval called what it is — a hazard a comment does not remove. The mutation
75
+ // probe patches this source in a disposable copy instead, and fails loudly if its
76
+ // anchor moves.
77
+ function bandOf(c) {
78
+ if (!c) return null;
79
+ const g = c.generation;
80
+ if (g && typeof g.mean === 'number') {
81
+ const sd = typeof g.sd === 'number' ? g.sd : 0;
82
+ return { mean: g.mean, sd, lo: g.mean - sd, hi: g.mean + sd, n: g.n_measured, source: 'generation' };
83
+ }
84
+ // v0.4 and earlier: the same measurement, under the names that version used.
85
+ // `score` is v0.1's spelling of the mean and is accepted for the same reason.
86
+ const mean = typeof c.mean === 'number' ? c.mean : (typeof c.score === 'number' ? c.score : null);
87
+ if (mean === null) return null;
88
+ const sd = typeof c.stddev === 'number' ? c.stddev : 0;
89
+ return { mean, sd, lo: mean - sd, hi: mean + sd, n: Array.isArray(c.samples) ? c.samples.length : null, source: 'legacy' };
90
+ }
91
+
92
+ function baselineReproduces(older, newer) {
93
+ const a = armOf(older, 'baseline');
94
+ const b = armOf(newer, 'baseline');
95
+ if (!a.length || !b.length) return { ok: false, key: 'baseline_missing' };
96
+ const bad = [];
97
+ for (const nb of b) {
98
+ const ob = a.find((x) => x.id === nb.id);
99
+ if (!ob) continue;
100
+ const bandA = bandOf(ob); const bandB = bandOf(nb);
101
+ if (!bandA || !bandB) return { ok: false, key: 'baseline_unmeasured' };
102
+ const overlap = bandA.lo <= bandB.hi && bandB.lo <= bandA.hi;
103
+ if (!overlap) bad.push({ id: nb.id, observed: round(bandB.mean), expected: round(bandA.mean) });
104
+ }
105
+ if (bad.length) return { ok: false, key: 'baseline_did_not_reproduce', ...bad[0], cases: bad };
106
+ return { ok: true };
107
+ }
108
+
109
+ // The verdict path. REFUSED is a RESULT, not an error: it carries no delta, and
110
+ // it is what #006 published three times.
111
+ function compare(older, newer) {
112
+ const pre = baselineReproduces(older, newer);
113
+ if (!pre.ok) {
114
+ const reason = REFUSAL_REASONS[pre.key]({ observed: pre.observed, expected: pre.expected });
115
+ return { verdict: 'REFUSED', reason, reason_key: pre.key, delta: null, cases: pre.cases || [] };
116
+ }
117
+ const ws = (r) => { const arm = armOf(r, 'with_skill'); const b = arm.map(bandOf).filter(Boolean); return b.length ? b.reduce((s, x) => s + x.mean, 0) / b.length : null; };
118
+ const a = ws(older); const b = ws(newer);
119
+ if (a === null || b === null) return { verdict: 'REFUSED', reason: REFUSAL_REASONS.baseline_unmeasured(), reason_key: 'baseline_unmeasured', delta: null };
120
+ return { verdict: 'MEASURED', delta: round(b - a), baseline_reproduced: true };
121
+ }
122
+
123
+ function triage(a, b) {
124
+ const g = (r, p, d) => p.split('.').reduce((o, k) => (o == null ? o : o[k]), r) ?? d;
125
+ const changed = (p) => g(a, p) !== g(b, p);
126
+ if (changed('run.model_id')) return { decision: 'rerun', reason: 'the model id differs, so the generation would differ' };
127
+ if (changed('suite.suite_hash')) return { decision: 'rerun', reason: 'the suite content differs, so the prompts would differ' };
128
+ if (changed('skill.content_hash')) return { decision: 'rerun', reason: 'the skill content differs, so the with-skill generation would differ' };
129
+ if (changed('run.judge.model_id')) return { decision: 'regrade', reason: 'only the judge differs, so the existing generations can be rescored' };
130
+ if (changed('run.rubric_hash')) return { decision: 'regrade', reason: 'only the rubric differs, so the existing generations can be rescored' };
131
+ return { decision: 'reuse', reason: 'model, suite, skill, judge and rubric all match, so the earlier result stands' };
132
+ }
133
+
134
+ module.exports = { compare, baselineReproduces, triage, REFUSAL_REASONS, bandOf };
@@ -0,0 +1,169 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ const { bandVerdict, round } = require('./stats');
5
+ const { bandOf } = require('./reuse');
6
+ const { EFFECT_FLOOR } = require('../config');
7
+
8
+ // Revision drift — the fifth report type (spec 009, Report #006).
9
+ //
10
+ // Every other report type holds the skill text fixed and moves something
11
+ // underneath it: the model release (#001, #003), the vendor surface (#002), the
12
+ // capability tier (#004), the axes and the price (#005). This one inverts the
13
+ // design. The substrate is held still — same model, same provider, same surface,
14
+ // same suite, same fixed judge, same sampling — and the SKILL'S OWN TEXT moves,
15
+ // from the revision a report pinned to the revision upstream ships today.
16
+ //
17
+ // This module holds the report type's LANGUAGE as code rather than as
18
+ // hand-written page copy. That is deliberate: the fairness rule below is the one
19
+ // a measurement project is most tempted to apply in one direction only, and a
20
+ // rule that lives in prose cannot be gated before the run that would tempt it.
21
+
22
+ // ── the cell headline ────────────────────────────────────────────────────────
23
+ // A summary of the per-case band-overlap verdicts, worded about the REVISION.
24
+ // The release-drift headline says "the skill is measurably weaker", which is a
25
+ // sentence about a skill under a moving model. Here the model is the control.
26
+ function revisionHeadline(perCase) {
27
+ const reg = perCase.filter((r) => r.verdict === 'regression').length;
28
+ const imp = perCase.filter((r) => r.verdict === 'improvement').length;
29
+ const s = (n) => (n === 1 ? '' : 's');
30
+ if (reg && imp) {
31
+ return `MIXED — the revision improved ${imp} case${s(imp)} and regressed ${reg} on non-overlapping bands.`;
32
+ }
33
+ if (reg) {
34
+ return `REVISION REGRESSED — ${reg} case${s(reg)} scored lower under the current upstream text (bands do not overlap).`;
35
+ }
36
+ if (imp) {
37
+ return `REVISION IMPROVED — ${imp} case${s(imp)} scored higher under the current upstream text (bands do not overlap); none regressed.`;
38
+ }
39
+ return 'WITHIN NOISE — the revision moved no case beyond its confidence band; the pinned text and the current text measure the same.';
40
+ }
41
+
42
+ // Classification word for a cell, from its headline. Kept separate so a caller
43
+ // can branch on the class without parsing prose.
44
+ function revisionClass(perCase) {
45
+ const h = revisionHeadline(perCase);
46
+ return h.split(' —')[0];
47
+ }
48
+
49
+ // ── the fairness sentence ────────────────────────────────────────────────────
50
+ // Spec 009 § Fairness, clauses 1 and 4. Where a revision IMPROVED a skill, the
51
+ // published #005 figure understates the pack a reader can install today; where it
52
+ // REGRESSED one, #005 overstates it. Both sentences are generated by the same
53
+ // function, from the same template, so the disclosure cannot quietly become a
54
+ // one-directional courtesy — the symmetry is a property of the code, and the
55
+ // gate asserts it.
56
+ //
57
+ // A cell within noise gets NO sentence. #005's figure stands unamended, because
58
+ // nothing was measured that would amend it, and manufacturing a hedge for a null
59
+ // result is how a report launders noise into a finding.
60
+ function fairnessSentence({ slug, classification, report005Delta, measuredDelta }) {
61
+ const cls = String(classification || '');
62
+ if (cls !== 'REVISION IMPROVED' && cls !== 'REVISION REGRESSED') return null;
63
+ const improved = cls === 'REVISION IMPROVED';
64
+ const direction = improved ? 'understates' : 'overstates';
65
+ const d = (n) => (n == null ? 'n/a' : (n >= 0 ? '+' : '') + Number(n).toFixed(3));
66
+ return `Report #005 measured ${slug} at ${d(report005Delta)} on the text it had pinned. `
67
+ + `Report #006 measures the current upstream revision at ${d(measuredDelta)} on the same substrate and the same suite. `
68
+ + `#005's published figure therefore ${direction} the pack upstream ships today for this skill, `
69
+ + `and is amended by this report rather than corrected in place.`;
70
+ }
71
+
72
+ // ── per-cell scoping disclosures ─────────────────────────────────────────────
73
+ // Spec 009 AC-10. One cell in Report #006 measures something other than what its
74
+ // upstream author changed it to do, and the reader looking at that row is the
75
+ // reader who needs to be told.
76
+ const SCOPING_NOTES = {
77
+ 'git-workflow-and-versioning':
78
+ 'This revision changes the frontmatter `description:` line. In a skill runtime a description is a '
79
+ + 'routing trigger: it decides whether the skill loads, and never reaches the model as guidance. '
80
+ + 'Driftproof makes no routing decision — it always injects the skill, and passes the whole file, '
81
+ + 'frontmatter included, as the system prompt. This cell therefore measures the revision as added '
82
+ + 'context and cannot measure it as a trigger.',
83
+ };
84
+ function scopingNote(slug) {
85
+ return SCOPING_NOTES[slug] || null;
86
+ }
87
+
88
+ // ── the baseline-reproduction control ────────────────────────────────────────
89
+ // Spec 009 AC-6, and the thing that makes the free pinned arm honest.
90
+ //
91
+ // Report #006 reuses #005's receipts as the pinned-text arm. That is valid only
92
+ // if the substrate has not moved, and `run.model_release_date` is null on every
93
+ // #005 receipt, so id equality is the only version evidence a receipt carries. A
94
+ // provider that re-points a concrete id at a new snapshot is invisible to it.
95
+ //
96
+ // It does not have to be. Every fresh run emits a BASELINE arm: the same cases,
97
+ // the same substrate, and no skill text at all. The revision cannot touch it by
98
+ // construction, so comparing the fresh baseline against the reused receipt's
99
+ // baseline re-measures exactly the thing id equality could not prove — at no
100
+ // extra cost, because that arm is already paid for.
101
+ //
102
+ // A cell whose baselines do not reproduce is NOT MEASURED. The reuse is a tested
103
+ // prediction, not an assumption the report asks the reader to grant.
104
+ function baselineBands(receipt) {
105
+ const out = {};
106
+ for (const c of receipt.results.cases) {
107
+ if (c.mode !== 'baseline') continue;
108
+ // ONE BAND DEFINITION (spec 016 AC-1). This built its own from the v0.4-shaped
109
+ // fields, which worked — and that is the point: three copies existed, two
110
+ // reading the legacy shape and one reading only v0.5, and the one that read
111
+ // only v0.5 was the one a cross-version control depended on (F-015-C). Routing
112
+ // every comparison path through `bandOf` means a future shape is added once.
113
+ const b = bandOf(c);
114
+ if (b) out[c.id] = { mean: b.mean, stddev: b.sd, source: b.source };
115
+ }
116
+ return out;
117
+ }
118
+
119
+ function baselineControl(reused, fresh) {
120
+ const A = baselineBands(reused);
121
+ const B = baselineBands(fresh);
122
+ const ids = [...new Set([...Object.keys(A), ...Object.keys(B)])];
123
+
124
+ const perCase = ids.map((id) => {
125
+ const before = A[id] || null;
126
+ const after = B[id] || null;
127
+ if (!before || !after) return { id, before, after, delta: null, moved: false, missing: true };
128
+ const delta = round(after.mean - before.mean);
129
+ // The same rule the study uses everywhere else: band separation AND the
130
+ // effect floor. A baseline that wobbles inside its band has not moved.
131
+ const raw = bandVerdict(before.mean, before.stddev, after.mean, after.stddev);
132
+ const separated = raw === 'regression' || raw === 'improvement';
133
+ return { id, before, after, delta, moved: separated && Math.abs(delta) >= EFFECT_FLOOR, missing: false };
134
+ });
135
+
136
+ const movedCases = perCase.filter((r) => r.moved);
137
+ const missing = perCase.filter((r) => r.missing);
138
+ const reproduced = movedCases.length === 0 && missing.length === 0;
139
+ const aggDelta = round(
140
+ (fresh.comparison && fresh.comparison.baseline_score != null ? fresh.comparison.baseline_score : 0)
141
+ - (reused.comparison && reused.comparison.baseline_score != null ? reused.comparison.baseline_score : 0),
142
+ );
143
+
144
+ return {
145
+ reproduced,
146
+ blocked: !reproduced,
147
+ verdict: reproduced ? 'MEASURED' : 'NOT MEASURED',
148
+ moved_cases: movedCases.map((r) => r.id),
149
+ missing_cases: missing.map((r) => r.id),
150
+ aggregate_baseline_delta: aggDelta,
151
+ floor: EFFECT_FLOOR,
152
+ // THE CONTROL PROVES NON-REPRODUCTION. IT CANNOT SAY WHY. These strings used
153
+ // to read 'the substrate moved' and 'the substrate held still' — a cause,
154
+ // asserted by a comparison that measures two baseline arms and nothing else.
155
+ // A 120-call stability probe then found generation-level sampling noise large
156
+ // enough to account for every gap this control saw, with no substrate movement
157
+ // required, and the report page retracted the claim while three committed
158
+ // control records still carried it (approval finding F-009-N). Reason strings
159
+ // only: no score, sample, hash or verdict changed with this edit.
160
+ reason: reproduced
161
+ ? 'the fresh baseline reproduces the reused receipt\'s baseline within the band and the floor, so the pinned-arm reuse stands for this cell'
162
+ : `the fresh baseline does not reproduce the reused receipt's baseline (${movedCases.length} case(s) moved beyond the band and the ${EFFECT_FLOOR} floor${missing.length ? `, ${missing.length} case(s) absent on one side` : ''}) — the reused pinned arm is not comparable to the fresh arm, so revision drift cannot be separated from whatever else changed in this cell; the control establishes non-reproduction and does not identify a cause`,
163
+ perCase,
164
+ };
165
+ }
166
+
167
+ module.exports = {
168
+ revisionHeadline, revisionClass, fairnessSentence, scopingNote, baselineControl,
169
+ };
package/lib/run.js CHANGED
@@ -1,7 +1,7 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
- const { complete, resolveModel, surfaceForModel, isMeteredSurface, CODEX_OVERHEAD_NOTE } = require('./provider');
4
+ const { complete, resolveModel, surfaceForModel, isMeteredSurface, retryPolicyForSurface, CODEX_OVERHEAD_NOTE } = require('./provider');
5
5
  const { gradeSamples, judgeSettings } = require('./judge');
6
6
  const { buildReceipt } = require('./receipt');
7
7
  const { sha256 } = require('./canonical');
@@ -11,7 +11,9 @@ const { runChecks } = require('./checks');
11
11
  const { estimateTokens } = require('./skillCost');
12
12
  const { hasUsage, normalizeUsage } = require('./usage');
13
13
  const { buildPricingSnapshot, computeEconomics } = require('./value');
14
- const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES } = require('../config');
14
+ const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES, DEV_MAX_CALLS } = require('../config');
15
+ const { SAMPLING, acrossDraws, nextAction } = require('./sampling');
16
+ const { suiteCanary } = require('./canary');
15
17
 
16
18
  // Known model release dates (best-effort; null when unknown). Recorded into the
17
19
  // receipt so drift reports can order runs by model age. Dateless model ids
@@ -36,9 +38,14 @@ function releaseDateFor(modelId) {
36
38
  }
37
39
 
38
40
  // Calls for one model run: each case does 2 generations (with_skill + baseline)
39
- // and 2×samples judge calls (each generation judged `samples` times).
40
- function projectCalls(caseCount, samples) {
41
- return caseCount * (2 + 2 * samples);
41
+ // and 2×samples judge calls (each generation judged `samples` times) — times the
42
+ // number of GENERATION DRAWS per arm (v0.5).
43
+ //
44
+ // `draws` DEFAULTS TO 1 so every existing caller projects exactly what it
45
+ // projected before; the runner's own cost guard passes the sampling MAXIMUM,
46
+ // which is the fail-safe direction for a guard that decides whether to spend.
47
+ function projectCalls(caseCount, samples, draws = 1) {
48
+ return caseCount * draws * (2 + 2 * samples);
42
49
  }
43
50
 
44
51
  // Ask the target model to perform one eval case. `withSkill` decides whether the
@@ -126,14 +133,45 @@ async function mapPool(items, concurrency, fn) {
126
133
  // keepTranscripts — when true, the run records transcripts:"retained-local"
127
134
  // and returns the raw generations + judge outputs so the
128
135
  // caller can write them to transcripts/<receipt-id>/.
136
+ // THE ONE PLACE A CALL TIMEOUT IS DECIDED (spec 017 AC-1, AC-2).
137
+ //
138
+ // PURE: a surface name and the run's options in, milliseconds out. No I/O, no
139
+ // clock, no provider — so a probe can ask what a run WOULD use without making a
140
+ // call, which is what makes this criterion assertable at all.
141
+ //
142
+ // An operator-supplied `opts.timeoutMs` still wins; what is gone is the silent
143
+ // literal that used to stand in for a policy. `Number.isFinite` rather than a
144
+ // truthiness test, so an explicit 0 is a value and not a fall-through.
145
+ // Declared as a NAMED FUNCTION EXPRESSION bound to a const, not as a bare
146
+ // declaration. The binding the module exports and the name inside the function
147
+ // are then separable, so a mutation probe can rename the inner function to prove
148
+ // the export is load-bearing without the module failing to load on an undefined
149
+ // identifier. The inner name is kept for stack traces.
150
+ const resolveCallTimeoutMs = function resolveCallTimeoutMs(surface, opts = {}) {
151
+ if (opts && Number.isFinite(opts.timeoutMs)) return opts.timeoutMs;
152
+ return retryPolicyForSurface(surface).timeoutMs;
153
+ };
154
+
129
155
  async function runSkillOnModel({ skill, model, opts = {} }) {
130
156
  const modelId = resolveModel(model);
131
157
  const judgeModel = opts.judgeModel ? resolveModel(opts.judgeModel) : modelId;
132
- const timeoutMs = opts.timeoutMs || 120000;
158
+ // THE SURFACE'S OWN DECLARED TIMEOUT, resolved per arm (spec 017 AC-1).
159
+ //
160
+ // This read `opts.timeoutMs || 120000`, and `lib/provider.js` documents that an
161
+ // explicit caller value wins — so the literal outranked the policy and the
162
+ // 300 s claude-cli timeout, written for cold-start-dominated CLI subprocesses,
163
+ // never executed. Report #007 lost 25 of 160 draws to
164
+ // `provider(claude-cli) timed out after 120000ms`, and one arm entirely.
165
+ //
166
+ // RESOLVED SEPARATELY FOR GENERATION AND JUDGE, because they can be different
167
+ // models on different surfaces: #007 generated on claude-fable-5 and judged on
168
+ // claude-haiku-4-5. One shared timeout would apply one surface's policy to both.
169
+ const genTimeoutMs = resolveCallTimeoutMs(surfaceForModel(modelId), opts);
170
+ const judgeTimeoutMs = resolveCallTimeoutMs(surfaceForModel(judgeModel), opts);
133
171
  // Optional per-case timeout overrides { caseId: ms }; a slow case can get a
134
172
  // longer budget without lengthening every other case's per-call timeout.
135
173
  const caseTimeoutMs = opts.caseTimeoutMs || {};
136
- const maxCalls = opts.maxCalls || 200;
174
+ const maxCalls = opts.maxCalls || DEV_MAX_CALLS;
137
175
  const samples = opts.samples || DEFAULT_JUDGE_SAMPLES;
138
176
  const concurrency = Math.max(1, opts.concurrency || 1);
139
177
  const onProgress = opts.onProgress || (() => {});
@@ -145,7 +183,10 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
145
183
 
146
184
  // Cost guard: project the whole run up front and refuse before spending a
147
185
  // single call if it would blow the cap.
148
- const projected = projectCalls(cases.length, samples);
186
+ // The WORST CASE, deliberately: escalation is adaptive and a guard that
187
+ // projects the floor would wave through a run that then draws ten times.
188
+ // Refusing a run that would have fit is recoverable; overspending is not.
189
+ const projected = projectCalls(cases.length, samples, SAMPLING.max);
149
190
  if (projected > maxCalls) {
150
191
  const e = new Error(`cost guard: projected ${projected} calls exceeds cap ${maxCalls} (${cases.length} cases × (2 + 2×${samples} samples)). Raise --max-calls or lower --max-cases/--samples.`);
151
192
  e.code = 'CALL_CAP';
@@ -163,43 +204,111 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
163
204
  const genKind = (ws) => (ws ? 'gen_with_skill' : 'gen_baseline');
164
205
  const pairs = await mapPool(tasks, concurrency, async ({ c, withSkill }) => {
165
206
  const mode = withSkill ? 'with_skill' : 'baseline';
166
- const ct = caseTimeoutMs[c.id] || timeoutMs;
167
- try {
168
- onProgress({ case: c.id, mode, phase: 'generate' });
169
- const gen = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs: ct });
170
- calls += 1;
171
- // Live budget: count the generation call INCLUDING retries, then hard-stop
172
- // if over 1.25× cap.
173
- if (budget) budget.add((gen.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
174
- const generationHash = sha256(String(gen.text || ''));
175
- onProgress({ case: c.id, mode, phase: 'judge', samples });
176
- const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ct, samples });
177
- // v0.4: the GENERATION call's usage is the skill-value measurement (the
178
- // judge's own usage rides separately on judge_usage, above).
179
- if (hasUsage(gen.usage)) jr.caseResult.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
180
- calls += samples;
181
- // Live budget: count all judge calls (retries included) for this (case, mode).
182
- if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judgeModel, 'judge'));
183
- onProgress({ case: c.id, mode, phase: 'done', outcome: jr.caseResult.outcome, score: jr.caseResult.mean, stddev: jr.caseResult.stddev });
184
- const transcript = keepTranscripts
185
- ? { id: c.id, mode, generation: String(gen.text || ''), judge_outputs: jr.sampleTexts }
186
- : null;
187
- return { caseResult: jr.caseResult, transcript };
188
- } catch (e) {
189
- if (e && e.code === 'BUDGET_HARDSTOP') throw e; // budget hard-stop stays fatal
190
- if (!isTimeout(e)) throw e; // non-timeout errors stay fatal
191
- // Persistent timeout → NON-FATAL: charge the consumed attempts, record the
192
- // case as failed_timeout (no fabricated samples), and continue the run.
193
- if (budget) {
194
- try {
195
- if (e.phase === 'judge') budget.add((e.judgeAttempts || 1) * perCallCostUSD(judgeModel, 'judge'));
196
- else budget.add((e.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
197
- } catch (be) { if (be && be.code === 'BUDGET_HARDSTOP') throw be; }
207
+ // A per-case override is an operator's explicit input and still wins, for
208
+ // both arms. Otherwise each arm uses its own surface's declared policy.
209
+ const ctGen = caseTimeoutMs[c.id] || genTimeoutMs;
210
+ const ctJudge = caseTimeoutMs[c.id] || judgeTimeoutMs;
211
+ // v0.5 — DRAW THE GENERATION n TIMES, and keep each draw's judge samples
212
+ // INSIDE that draw. Pooling k×n scores into one list is precisely what made
213
+ // generation noise read as judge noise: it is the defect Report #006 exists
214
+ // to name, and the nesting is the whole measurement.
215
+ const draws = [];
216
+ let last = null; // last MEASURED draw — carries the v0.4-shaped fields
217
+ let lastTranscript = null;
218
+ let action = { stop: false, reason: 'below_min' };
219
+ let fatal = null;
220
+
221
+ while (!action.stop && draws.length < SAMPLING.max) {
222
+ const drawIndex = draws.length;
223
+ try {
224
+ onProgress({ case: c.id, mode, phase: 'generate', draw: drawIndex });
225
+ const gen = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs: ctGen });
226
+ calls += 1;
227
+ if (budget) budget.add((gen.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
228
+ const generationHash = sha256(String(gen.text || ''));
229
+ onProgress({ case: c.id, mode, phase: 'judge', samples, draw: drawIndex });
230
+ const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ctJudge, samples });
231
+ calls += samples;
232
+ if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judgeModel, 'judge'));
233
+ const draw = {
234
+ draw_index: drawIndex,
235
+ generation_hash: generationHash,
236
+ status: 'measured',
237
+ samples: jr.caseResult.samples,
238
+ judge_sample_hashes: jr.caseResult.judge_sample_hashes,
239
+ mean: jr.caseResult.mean,
240
+ stddev: jr.caseResult.stddev,
241
+ };
242
+ if (hasUsage(gen.usage)) draw.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
243
+ if (jr.caseResult.judge_usage) draw.judge_usage = jr.caseResult.judge_usage;
244
+ draws.push(draw);
245
+ last = jr;
246
+ if (keepTranscripts) lastTranscript = { id: c.id, mode, generation: String(gen.text || ''), judge_outputs: jr.sampleTexts };
247
+ } catch (e) {
248
+ if (e && e.code === 'BUDGET_HARDSTOP') throw e; // budget hard-stop stays fatal
249
+ if (!isTimeout(e)) { fatal = e; break; } // non-timeout errors stay fatal
250
+ if (budget) {
251
+ try {
252
+ if (e.phase === 'judge') budget.add((e.judgeAttempts || 1) * perCallCostUSD(judgeModel, 'judge'));
253
+ else budget.add((e.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
254
+ } catch (be) { if (be && be.code === 'BUDGET_HARDSTOP') throw be; }
255
+ }
256
+ // F-009-L: the draw is UNMEASURED. No score, no fabricated samples, and
257
+ // it is excluded from every statistic rather than counted as a zero — a
258
+ // zero asserts a measurement, and a timeout is the absence of one.
259
+ draws.push({
260
+ draw_index: drawIndex,
261
+ generation_hash: null,
262
+ status: 'unmeasured',
263
+ reason: String((e && e.message) || 'timeout').slice(0, 200),
264
+ samples: [],
265
+ mean: null,
266
+ stddev: null,
267
+ });
268
+ onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: String((e && e.message) || 'timeout') });
198
269
  }
270
+ action = nextAction(draws);
271
+ }
272
+ if (fatal) throw fatal;
273
+
274
+ const agg = acrossDraws(draws);
275
+ const generation = {
276
+ n_planned: SAMPLING.min,
277
+ n_drawn: agg.n_drawn,
278
+ n_measured: agg.n_measured,
279
+ n_unmeasured: agg.n_unmeasured,
280
+ stopping_reason: action.reason,
281
+ mean: agg.mean,
282
+ sd: agg.sd,
283
+ judge_sd_mean: agg.judge_sd_mean,
284
+ variance_ratio: agg.variance_ratio,
285
+ // WHICH null, when it is null (F-014-C). Copied through explicitly rather
286
+ // than spread from `agg`: this assembly names its keys one by one, and the
287
+ // canary was dropped by exactly such an assembly silently gaining a field
288
+ // upstream that nothing here carried down (F-014-D).
289
+ variance_ratio_unavailable: agg.variance_ratio_unavailable,
290
+ draws,
291
+ };
292
+
293
+ // Every draw failed: the case is recorded failed_timeout as before, but it
294
+ // now carries the draw list showing WHAT failed and how often.
295
+ if (!last) {
199
296
  failedCases += 1;
200
- onProgress({ case: c.id, mode, phase: 'failed', reason: String((e && e.message) || 'timeout') });
201
- return { caseResult: { id: c.id, mode, case_status: 'failed_timeout', reason: String((e && e.message) || 'timeout').slice(0, 200) }, transcript: null };
297
+ return { caseResult: { id: c.id, mode, case_status: 'failed_timeout', reason: (draws[draws.length - 1] || {}).reason || 'timeout', generation }, transcript: null };
202
298
  }
299
+
300
+ // The v0.4-shaped fields now describe the DRAW SET, not one arbitrary draw,
301
+ // so a v0.4 reader pointed at a v0.5 receipt reads the aggregate rather than
302
+ // whichever draw happened to be last. `generation_hash`, `samples` and
303
+ // `judge_sample_hashes` continue to describe the last measured draw, which
304
+ // is the one they have always described; RECEIPT.md states this.
305
+ const caseResult = { ...last.caseResult, generation };
306
+ caseResult.mean = agg.mean;
307
+ caseResult.score = agg.mean;
308
+ caseResult.stddev = agg.sd;
309
+ caseResult.outcome = outcomeFor(agg.mean, agg.sd, c.pass_threshold);
310
+ onProgress({ case: c.id, mode, phase: 'done', outcome: caseResult.outcome, score: caseResult.mean, stddev: caseResult.stddev });
311
+ return { caseResult, transcript: lastTranscript };
203
312
  });
204
313
  const caseResults = pairs.map((p) => p.caseResult);
205
314
  const transcripts = keepTranscripts ? pairs.map((p) => p.transcript) : null;
@@ -229,7 +338,18 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
229
338
  // v0.3.1 value-per-token axis: estimated SKILL.md token size.
230
339
  tokens: estimateTokens(skill.skillMd),
231
340
  },
232
- suite: { format: skill.suite.format, suiteHash: skill.suite.suiteHash, caseCount: skill.suite.caseCount },
341
+ // v0.5: the suite canary. Derived from the suite identity and its case ids,
342
+ // so it is stable without a registry and distinct across suites — a leaked
343
+ // suite is detectable in a corpus. A detection aid, not a control.
344
+ suite: {
345
+ format: skill.suite.format,
346
+ suiteHash: skill.suite.suiteHash,
347
+ caseCount: skill.suite.caseCount,
348
+ // The FULL suite, never the post---max-cases list: a canary derived from a
349
+ // truncated run is not stable for the suite, and a canary that moves
350
+ // cannot say which suite leaked. Approval finding, spec 014.
351
+ canary: suiteCanary({ id: skill.name || skill.suite.suiteHash, cases: (skill.suite.cases || cases).map((c) => ({ id: c.id })) }),
352
+ },
233
353
  run: {
234
354
  model_id: modelId,
235
355
  model_release_date: releaseDateFor(modelId),
@@ -277,7 +397,7 @@ function summarizeReceipt(receipt) {
277
397
  const aggs = receipt.results.aggregates;
278
398
  const sign = cmp.delta >= 0 ? '+' : '';
279
399
  if (receipt.run.status === 'incomplete') {
280
- L.push(`> ⏱ **INCOMPLETE** — ${receipt.run.failed_case_count} case(s) failed (timed out after retries) and are EXCLUDED from the aggregates below; this receipt must not be used to compute a drift/durability verdict.`);
400
+ L.push(`> ⏱ **INCOMPLETE** — ${receipt.run.failed_case_count} case(s) had an arm that could not be measured and are EXCLUDED from the aggregates below, BOTH arms together; this receipt must not be used to compute a drift/durability verdict.`);
281
401
  L.push('');
282
402
  }
283
403
  L.push(`with_skill **${band(aggs.with_skill.mean_score, aggs.with_skill.stddev)}** vs baseline **${band(aggs.baseline.mean_score, aggs.baseline.stddev)}**`);
@@ -300,4 +420,4 @@ function summarizeReceipt(receipt) {
300
420
  return L.join('\n');
301
421
  }
302
422
 
303
- module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor };
423
+ module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor, resolveCallTimeoutMs };