driftproof 0.6.0 → 0.7.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/receipt.js CHANGED
@@ -15,7 +15,12 @@ const SCHEMA_FILES = {
15
15
  '0.2': 'receipt.v0.2.schema.json',
16
16
  '0.3': 'receipt.v0.3.schema.json',
17
17
  '0.3.1': 'receipt.v0.3.1.schema.json',
18
- '0.4': 'receipt.schema.json',
18
+ // v0.4 moved from the unversioned filename to a version-pinned one when v0.5
19
+ // took the current pointer. Without this the archive would lose v0.4: every
20
+ // published v0.4 receipt asserts conformance by NUMBER, and the number has to
21
+ // keep resolving to the schema it meant (AC-10).
22
+ '0.4': 'receipt.v0.4.schema.json',
23
+ '0.5': 'receipt.schema.json',
19
24
  };
20
25
 
21
26
  const _validators = {};
@@ -84,15 +89,70 @@ function aggregate(caseResults) {
84
89
  function buildReceipt({ skill, suite, run, cases, economics = null, verificationLevel = 'TESTED', editorialReviews = null }) {
85
90
  // v0.3.1: cases marked failed_timeout are recorded in results.cases but EXCLUDED
86
91
  // from aggregates — a band is never fabricated from a case that did not complete.
87
- const okCases = cases.filter((c) => c.case_status !== 'failed_timeout');
88
- const failedCount = cases.length - okCases.length;
92
+ //
93
+ // PAIRWISE, NOT PER ARM (spec 017 AC-5, AC-6). This filtered a FLAT list and
94
+ // then split by mode, so a case whose baseline failed kept its with_skill arm:
95
+ // #007's cell 3 recorded with_skill case_count 7 against baseline 6, and its
96
+ // headline `delta` of +0.116 was a 7-case mean minus a 6-case mean.
97
+ //
98
+ // `comparison.delta` is PAIRED BY CONSTRUCTION — one suite, measured twice —
99
+ // and a paired statistic computed over unequal sets is not the statistic it
100
+ // names. So an arm that cannot be measured removes its CASE from both sides.
101
+ //
102
+ // Exclusion, not refusal, and the reason is recorded rather than argued: an
103
+ // aggregate is a summary statistic, not a verdict, so the "refuse rather than
104
+ // assert" rule that governs verdicts does not reach it; and refusing the whole
105
+ // aggregate would discard thirteen sound arms because one failed. What the
106
+ // aggregate owes a reader instead is that it says what it covered, which
107
+ // `excluded_cases` provides.
108
+ //
109
+ // The excluded case STAYS in `results.cases`. It is removed from the mean, not
110
+ // from the record — deleting the evidence of a failure is a different and worse
111
+ // defect than averaging over it.
112
+ const armUnusable = (c) => c.case_status === 'failed_timeout'
113
+ || (c.mean == null && c.score == null);
114
+ const excludedIds = new Map();
115
+ for (const c of cases) {
116
+ if (!armUnusable(c)) continue;
117
+ if (excludedIds.has(c.id)) { excludedIds.get(c.id).modes.push(c.mode); continue; }
118
+ excludedIds.set(c.id, {
119
+ id: c.id,
120
+ modes: [c.mode],
121
+ reason: c.reason
122
+ || (c.generation && c.generation.stopping_reason === 'unmeasured_exhausted'
123
+ ? 'every generation draw was unmeasured'
124
+ : 'the arm has no measured result'),
125
+ });
126
+ }
127
+ const okCases = cases.filter((c) => !excludedIds.has(c.id));
89
128
  const withSkill = okCases.filter((c) => c.mode === 'with_skill');
90
129
  const baseline = okCases.filter((c) => c.mode === 'baseline');
130
+ const excludedCases = [...excludedIds.values()];
131
+ // `run.failed_case_count` COUNTS CASES, and is computed FROM the list it
132
+ // summarises rather than alongside it, so the two fields cannot disagree.
133
+ //
134
+ // It was `cases.length - okCases.length`, which counted ROWS. That was the
135
+ // same number while exclusion was per arm; once AC-5 made exclusion pairwise
136
+ // the subtraction removed BOTH arms of every excluded case, so one failed arm
137
+ // reported 2. Measured on the archive before this fix: #006's writing-plans
138
+ // receipt recomputed to 4 against the 2 it records, and #007's to 2 against 1
139
+ // — two published figures doubled by a change that never named this field.
140
+ //
141
+ // The field's name says cases and the aggregates exclude by case, so the
142
+ // count is the length of `results.aggregates.excluded_cases` and nothing
143
+ // else. A case whose BOTH arms failed is one exclusion and counts once.
144
+ const failedCount = excludedCases.length;
91
145
  const aggWith = aggregate(withSkill);
92
146
  const aggBase = aggregate(baseline);
93
147
 
94
148
  const receipt = {
95
149
  schema_version: RECEIPT_SCHEMA_VERSION,
150
+ // v0.5 CAPABILITY FLAG (F-014-F). DERIVED FROM THE CASES THEMSELVES, never
151
+ // from a caller's argument: the exposure this closes is a receipt asserting
152
+ // something it does not carry, and a flag taken on trust from the caller
153
+ // would be the same defect with an extra step. Absent when no case carries a
154
+ // draw set, which is what keeps every legacy and imported receipt valid.
155
+ ...(cases.some((c) => c && c.generation) ? { generation_sampled: true } : {}),
96
156
  skill: {
97
157
  name: skill.name,
98
158
  version: skill.version,
@@ -102,6 +162,12 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
102
162
  format: suite.format,
103
163
  suite_hash: suite.suiteHash,
104
164
  case_count: suite.caseCount,
165
+ // v0.5: the per-suite canary. THIS is the canonical assembly — the runner
166
+ // built the field and this function dropped it, so the live smoke emitted
167
+ // `canary: undefined` and the schema, which makes it optional, said
168
+ // nothing. Caught by reading the receipt a real run produced, not by a
169
+ // gate; the assertion that would have caught it is added with the fix.
170
+ ...(suite.canary ? { canary: suite.canary } : {}),
105
171
  },
106
172
  run: {
107
173
  model_id: run.model_id,
@@ -120,7 +186,13 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
120
186
  },
121
187
  results: {
122
188
  cases,
123
- aggregates: { with_skill: aggWith, baseline: aggBase },
189
+ aggregates: {
190
+ with_skill: aggWith,
191
+ baseline: aggBase,
192
+ // Present only when something was excluded, so a clean run's receipt is
193
+ // unchanged and the archive does not acquire an empty field.
194
+ ...(excludedCases.length ? { excluded_cases: excludedCases } : {}),
195
+ },
124
196
  },
125
197
  comparison: {
126
198
  with_skill_score: aggWith.mean_score,
package/lib/reuse.js ADDED
@@ -0,0 +1,134 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // Rerun / regrade / reuse, and the baseline-reproduction precondition.
5
+ // Receipt spec v0.5.
6
+ //
7
+ // TWO THINGS LIVE HERE, both decided from what two receipts RECORD rather than
8
+ // from a flag someone passed:
9
+ //
10
+ // 1. THE PRECONDITION (AC-6). Report #006's whole result. The no-skill arm
11
+ // contains no skill text, so a skill revision cannot move it. If it fails
12
+ // to reproduce the arm an earlier receipt measured on the same model, the
13
+ // same surface and the same suite, then whatever else changed, the two
14
+ // receipts are not measuring the same thing — and a verdict computed across
15
+ // them would be a comparison whose premise was never checked. That is the
16
+ // F-009-X class. The comparison is REFUSED and no verdict is asserted.
17
+ //
18
+ // 2. THE TRIAGE (AC-8). Terminal-Bench 4.0 distinguishes rerunning a task,
19
+ // regrading an existing transcript, and reusing a prior result. Adopted,
20
+ // translated: our unit is a (case, arm) draw set, and the trigger is the
21
+ // receipt's own recorded provenance — model, suite and skill content force
22
+ // a rerun because the generation would differ; judge and rubric force a
23
+ // regrade because only the scoring would; neither changing permits reuse.
24
+ //
25
+ // PURE: two receipts in, a decision out. No I/O, no provider. F-009-K's lesson
26
+ // applied — the decision derives from the artifact, never from a filename.
27
+
28
+ const { round } = require('./stats');
29
+
30
+ // EVERY REASON STATES WHAT WAS OBSERVED AND STOPS THERE (F-009-N). The control
31
+ // proves non-reproduction; it cannot say why. A reason that named a cause — "the
32
+ // skill regressed", "the model got worse" — would be a finding this instrument
33
+ // did not measure, which is the one thing it must never publish.
34
+ const REFUSAL_REASONS = {
35
+ baseline_did_not_reproduce: ({ observed, expected } = {}) =>
36
+ `the baseline arm did not reproduce: this run observed ${observed}, the earlier receipt recorded ${expected}, and the bands do not overlap. No verdict is asserted; the control shows non-reproduction and cannot establish a cause.`,
37
+ baseline_missing: () =>
38
+ 'no baseline arm is present in one of the two receipts, so reproduction cannot be checked and no verdict is asserted.',
39
+ baseline_unmeasured: () =>
40
+ 'the baseline arm has no measured draws, so reproduction cannot be checked and no verdict is asserted.',
41
+ incomparable_surface: () =>
42
+ 'the two receipts record different surfaces, so the earlier measurement cannot stand as a control here and no verdict is asserted.',
43
+ };
44
+
45
+ const casesOf = (r) => ((r && r.results && r.results.cases) || []);
46
+ const armOf = (r, mode) => casesOf(r).filter((c) => c.mode === mode);
47
+
48
+ // THE ONE BAND DEFINITION, and it resolves the shape the receipt actually
49
+ // recorded (spec 016 AC-1, closing F-015-C).
50
+ //
51
+ // WHAT WENT WRONG. This read `c.generation.mean` and nothing else. That block
52
+ // exists on v0.5 receipts and on no earlier one — v0.4 records the same
53
+ // measurement as `c.mean` and `c.stddev` — so every v0.4 case resolved to `null`
54
+ // and `baselineReproduces` returned `baseline_unmeasured` for EVERY v0.4→v0.5
55
+ // pair, always, whatever the baselines had done. The precondition written
56
+ // specifically for cross-version comparison could not pass against the archive
57
+ // it exists to compare against, and Report #007's entire comparison step is
58
+ // v0.4-against-v0.5.
59
+ //
60
+ // It survived 487 repo assertions and three approval rounds because every
61
+ // assertion that ever exercised it handed it a fixture built to the CURRENT
62
+ // schema. That is the seventh narrowing class, `fixture-vs-real-artifact`.
63
+ //
64
+ // THE BAND CARRIES WHICH SHAPE IT CAME FROM, and the two are not the same
65
+ // statistic: v0.5's `sd` is spread ACROSS DRAWS, v0.4's `stddev` is spread across
66
+ // JUDGE SAMPLES of one draw. Comparing them is the only comparison v0.4 admits —
67
+ // it is the band that version measured — but a reader is owed the fact that the
68
+ // older side is a judge-level band, so `source` travels with it and the differ
69
+ // says so rather than presenting them as like for like.
70
+ //
71
+ // NO SWITCH GUARDS THE LEGACY PATH. An earlier revision carried a const-true
72
+ // `LEGACY_BAND_ENABLED` so a probe could disable the fallback without editing the
73
+ // logic; that leaves the pre-fix behaviour resident in the shipped tree, which an
74
+ // approval called what it is — a hazard a comment does not remove. The mutation
75
+ // probe patches this source in a disposable copy instead, and fails loudly if its
76
+ // anchor moves.
77
+ function bandOf(c) {
78
+ if (!c) return null;
79
+ const g = c.generation;
80
+ if (g && typeof g.mean === 'number') {
81
+ const sd = typeof g.sd === 'number' ? g.sd : 0;
82
+ return { mean: g.mean, sd, lo: g.mean - sd, hi: g.mean + sd, n: g.n_measured, source: 'generation' };
83
+ }
84
+ // v0.4 and earlier: the same measurement, under the names that version used.
85
+ // `score` is v0.1's spelling of the mean and is accepted for the same reason.
86
+ const mean = typeof c.mean === 'number' ? c.mean : (typeof c.score === 'number' ? c.score : null);
87
+ if (mean === null) return null;
88
+ const sd = typeof c.stddev === 'number' ? c.stddev : 0;
89
+ return { mean, sd, lo: mean - sd, hi: mean + sd, n: Array.isArray(c.samples) ? c.samples.length : null, source: 'legacy' };
90
+ }
91
+
92
+ function baselineReproduces(older, newer) {
93
+ const a = armOf(older, 'baseline');
94
+ const b = armOf(newer, 'baseline');
95
+ if (!a.length || !b.length) return { ok: false, key: 'baseline_missing' };
96
+ const bad = [];
97
+ for (const nb of b) {
98
+ const ob = a.find((x) => x.id === nb.id);
99
+ if (!ob) continue;
100
+ const bandA = bandOf(ob); const bandB = bandOf(nb);
101
+ if (!bandA || !bandB) return { ok: false, key: 'baseline_unmeasured' };
102
+ const overlap = bandA.lo <= bandB.hi && bandB.lo <= bandA.hi;
103
+ if (!overlap) bad.push({ id: nb.id, observed: round(bandB.mean), expected: round(bandA.mean) });
104
+ }
105
+ if (bad.length) return { ok: false, key: 'baseline_did_not_reproduce', ...bad[0], cases: bad };
106
+ return { ok: true };
107
+ }
108
+
109
+ // The verdict path. REFUSED is a RESULT, not an error: it carries no delta, and
110
+ // it is what #006 published three times.
111
+ function compare(older, newer) {
112
+ const pre = baselineReproduces(older, newer);
113
+ if (!pre.ok) {
114
+ const reason = REFUSAL_REASONS[pre.key]({ observed: pre.observed, expected: pre.expected });
115
+ return { verdict: 'REFUSED', reason, reason_key: pre.key, delta: null, cases: pre.cases || [] };
116
+ }
117
+ const ws = (r) => { const arm = armOf(r, 'with_skill'); const b = arm.map(bandOf).filter(Boolean); return b.length ? b.reduce((s, x) => s + x.mean, 0) / b.length : null; };
118
+ const a = ws(older); const b = ws(newer);
119
+ if (a === null || b === null) return { verdict: 'REFUSED', reason: REFUSAL_REASONS.baseline_unmeasured(), reason_key: 'baseline_unmeasured', delta: null };
120
+ return { verdict: 'MEASURED', delta: round(b - a), baseline_reproduced: true };
121
+ }
122
+
123
+ function triage(a, b) {
124
+ const g = (r, p, d) => p.split('.').reduce((o, k) => (o == null ? o : o[k]), r) ?? d;
125
+ const changed = (p) => g(a, p) !== g(b, p);
126
+ if (changed('run.model_id')) return { decision: 'rerun', reason: 'the model id differs, so the generation would differ' };
127
+ if (changed('suite.suite_hash')) return { decision: 'rerun', reason: 'the suite content differs, so the prompts would differ' };
128
+ if (changed('skill.content_hash')) return { decision: 'rerun', reason: 'the skill content differs, so the with-skill generation would differ' };
129
+ if (changed('run.judge.model_id')) return { decision: 'regrade', reason: 'only the judge differs, so the existing generations can be rescored' };
130
+ if (changed('run.rubric_hash')) return { decision: 'regrade', reason: 'only the rubric differs, so the existing generations can be rescored' };
131
+ return { decision: 'reuse', reason: 'model, suite, skill, judge and rubric all match, so the earlier result stands' };
132
+ }
133
+
134
+ module.exports = { compare, baselineReproduces, triage, REFUSAL_REASONS, bandOf };
package/lib/revision.js CHANGED
@@ -2,6 +2,7 @@
2
2
  'use strict';
3
3
 
4
4
  const { bandVerdict, round } = require('./stats');
5
+ const { bandOf } = require('./reuse');
5
6
  const { EFFECT_FLOOR } = require('../config');
6
7
 
7
8
  // Revision drift — the fifth report type (spec 009, Report #006).
@@ -104,7 +105,13 @@ function baselineBands(receipt) {
104
105
  const out = {};
105
106
  for (const c of receipt.results.cases) {
106
107
  if (c.mode !== 'baseline') continue;
107
- out[c.id] = { mean: c.mean != null ? c.mean : c.score, stddev: c.stddev || 0 };
108
+ // ONE BAND DEFINITION (spec 016 AC-1). This built its own from the v0.4-shaped
109
+ // fields, which worked — and that is the point: three copies existed, two
110
+ // reading the legacy shape and one reading only v0.5, and the one that read
111
+ // only v0.5 was the one a cross-version control depended on (F-015-C). Routing
112
+ // every comparison path through `bandOf` means a future shape is added once.
113
+ const b = bandOf(c);
114
+ if (b) out[c.id] = { mean: b.mean, stddev: b.sd, source: b.source };
108
115
  }
109
116
  return out;
110
117
  }
package/lib/run.js CHANGED
@@ -1,7 +1,7 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
- const { complete, resolveModel, surfaceForModel, isMeteredSurface, CODEX_OVERHEAD_NOTE } = require('./provider');
4
+ const { complete, resolveModel, surfaceForModel, isMeteredSurface, retryPolicyForSurface, CODEX_OVERHEAD_NOTE } = require('./provider');
5
5
  const { gradeSamples, judgeSettings } = require('./judge');
6
6
  const { buildReceipt } = require('./receipt');
7
7
  const { sha256 } = require('./canonical');
@@ -11,7 +11,9 @@ const { runChecks } = require('./checks');
11
11
  const { estimateTokens } = require('./skillCost');
12
12
  const { hasUsage, normalizeUsage } = require('./usage');
13
13
  const { buildPricingSnapshot, computeEconomics } = require('./value');
14
- const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES } = require('../config');
14
+ const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES, DEV_MAX_CALLS } = require('../config');
15
+ const { SAMPLING, acrossDraws, nextAction } = require('./sampling');
16
+ const { suiteCanary } = require('./canary');
15
17
 
16
18
  // Known model release dates (best-effort; null when unknown). Recorded into the
17
19
  // receipt so drift reports can order runs by model age. Dateless model ids
@@ -36,9 +38,14 @@ function releaseDateFor(modelId) {
36
38
  }
37
39
 
38
40
  // Calls for one model run: each case does 2 generations (with_skill + baseline)
39
- // and 2×samples judge calls (each generation judged `samples` times).
40
- function projectCalls(caseCount, samples) {
41
- return caseCount * (2 + 2 * samples);
41
+ // and 2×samples judge calls (each generation judged `samples` times) — times the
42
+ // number of GENERATION DRAWS per arm (v0.5).
43
+ //
44
+ // `draws` DEFAULTS TO 1 so every existing caller projects exactly what it
45
+ // projected before; the runner's own cost guard passes the sampling MAXIMUM,
46
+ // which is the fail-safe direction for a guard that decides whether to spend.
47
+ function projectCalls(caseCount, samples, draws = 1) {
48
+ return caseCount * draws * (2 + 2 * samples);
42
49
  }
43
50
 
44
51
  // Ask the target model to perform one eval case. `withSkill` decides whether the
@@ -126,14 +133,45 @@ async function mapPool(items, concurrency, fn) {
126
133
  // keepTranscripts — when true, the run records transcripts:"retained-local"
127
134
  // and returns the raw generations + judge outputs so the
128
135
  // caller can write them to transcripts/<receipt-id>/.
136
+ // THE ONE PLACE A CALL TIMEOUT IS DECIDED (spec 017 AC-1, AC-2).
137
+ //
138
+ // PURE: a surface name and the run's options in, milliseconds out. No I/O, no
139
+ // clock, no provider — so a probe can ask what a run WOULD use without making a
140
+ // call, which is what makes this criterion assertable at all.
141
+ //
142
+ // An operator-supplied `opts.timeoutMs` still wins; what is gone is the silent
143
+ // literal that used to stand in for a policy. `Number.isFinite` rather than a
144
+ // truthiness test, so an explicit 0 is a value and not a fall-through.
145
+ // Declared as a NAMED FUNCTION EXPRESSION bound to a const, not as a bare
146
+ // declaration. The binding the module exports and the name inside the function
147
+ // are then separable, so a mutation probe can rename the inner function to prove
148
+ // the export is load-bearing without the module failing to load on an undefined
149
+ // identifier. The inner name is kept for stack traces.
150
+ const resolveCallTimeoutMs = function resolveCallTimeoutMs(surface, opts = {}) {
151
+ if (opts && Number.isFinite(opts.timeoutMs)) return opts.timeoutMs;
152
+ return retryPolicyForSurface(surface).timeoutMs;
153
+ };
154
+
129
155
  async function runSkillOnModel({ skill, model, opts = {} }) {
130
156
  const modelId = resolveModel(model);
131
157
  const judgeModel = opts.judgeModel ? resolveModel(opts.judgeModel) : modelId;
132
- const timeoutMs = opts.timeoutMs || 120000;
158
+ // THE SURFACE'S OWN DECLARED TIMEOUT, resolved per arm (spec 017 AC-1).
159
+ //
160
+ // This read `opts.timeoutMs || 120000`, and `lib/provider.js` documents that an
161
+ // explicit caller value wins — so the literal outranked the policy and the
162
+ // 300 s claude-cli timeout, written for cold-start-dominated CLI subprocesses,
163
+ // never executed. Report #007 lost 25 of 160 draws to
164
+ // `provider(claude-cli) timed out after 120000ms`, and one arm entirely.
165
+ //
166
+ // RESOLVED SEPARATELY FOR GENERATION AND JUDGE, because they can be different
167
+ // models on different surfaces: #007 generated on claude-fable-5 and judged on
168
+ // claude-haiku-4-5. One shared timeout would apply one surface's policy to both.
169
+ const genTimeoutMs = resolveCallTimeoutMs(surfaceForModel(modelId), opts);
170
+ const judgeTimeoutMs = resolveCallTimeoutMs(surfaceForModel(judgeModel), opts);
133
171
  // Optional per-case timeout overrides { caseId: ms }; a slow case can get a
134
172
  // longer budget without lengthening every other case's per-call timeout.
135
173
  const caseTimeoutMs = opts.caseTimeoutMs || {};
136
- const maxCalls = opts.maxCalls || 200;
174
+ const maxCalls = opts.maxCalls || DEV_MAX_CALLS;
137
175
  const samples = opts.samples || DEFAULT_JUDGE_SAMPLES;
138
176
  const concurrency = Math.max(1, opts.concurrency || 1);
139
177
  const onProgress = opts.onProgress || (() => {});
@@ -145,7 +183,10 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
145
183
 
146
184
  // Cost guard: project the whole run up front and refuse before spending a
147
185
  // single call if it would blow the cap.
148
- const projected = projectCalls(cases.length, samples);
186
+ // The WORST CASE, deliberately: escalation is adaptive and a guard that
187
+ // projects the floor would wave through a run that then draws ten times.
188
+ // Refusing a run that would have fit is recoverable; overspending is not.
189
+ const projected = projectCalls(cases.length, samples, SAMPLING.max);
149
190
  if (projected > maxCalls) {
150
191
  const e = new Error(`cost guard: projected ${projected} calls exceeds cap ${maxCalls} (${cases.length} cases × (2 + 2×${samples} samples)). Raise --max-calls or lower --max-cases/--samples.`);
151
192
  e.code = 'CALL_CAP';
@@ -163,43 +204,111 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
163
204
  const genKind = (ws) => (ws ? 'gen_with_skill' : 'gen_baseline');
164
205
  const pairs = await mapPool(tasks, concurrency, async ({ c, withSkill }) => {
165
206
  const mode = withSkill ? 'with_skill' : 'baseline';
166
- const ct = caseTimeoutMs[c.id] || timeoutMs;
167
- try {
168
- onProgress({ case: c.id, mode, phase: 'generate' });
169
- const gen = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs: ct });
170
- calls += 1;
171
- // Live budget: count the generation call INCLUDING retries, then hard-stop
172
- // if over 1.25× cap.
173
- if (budget) budget.add((gen.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
174
- const generationHash = sha256(String(gen.text || ''));
175
- onProgress({ case: c.id, mode, phase: 'judge', samples });
176
- const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ct, samples });
177
- // v0.4: the GENERATION call's usage is the skill-value measurement (the
178
- // judge's own usage rides separately on judge_usage, above).
179
- if (hasUsage(gen.usage)) jr.caseResult.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
180
- calls += samples;
181
- // Live budget: count all judge calls (retries included) for this (case, mode).
182
- if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judgeModel, 'judge'));
183
- onProgress({ case: c.id, mode, phase: 'done', outcome: jr.caseResult.outcome, score: jr.caseResult.mean, stddev: jr.caseResult.stddev });
184
- const transcript = keepTranscripts
185
- ? { id: c.id, mode, generation: String(gen.text || ''), judge_outputs: jr.sampleTexts }
186
- : null;
187
- return { caseResult: jr.caseResult, transcript };
188
- } catch (e) {
189
- if (e && e.code === 'BUDGET_HARDSTOP') throw e; // budget hard-stop stays fatal
190
- if (!isTimeout(e)) throw e; // non-timeout errors stay fatal
191
- // Persistent timeout → NON-FATAL: charge the consumed attempts, record the
192
- // case as failed_timeout (no fabricated samples), and continue the run.
193
- if (budget) {
194
- try {
195
- if (e.phase === 'judge') budget.add((e.judgeAttempts || 1) * perCallCostUSD(judgeModel, 'judge'));
196
- else budget.add((e.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
197
- } catch (be) { if (be && be.code === 'BUDGET_HARDSTOP') throw be; }
207
+ // A per-case override is an operator's explicit input and still wins, for
208
+ // both arms. Otherwise each arm uses its own surface's declared policy.
209
+ const ctGen = caseTimeoutMs[c.id] || genTimeoutMs;
210
+ const ctJudge = caseTimeoutMs[c.id] || judgeTimeoutMs;
211
+ // v0.5 — DRAW THE GENERATION n TIMES, and keep each draw's judge samples
212
+ // INSIDE that draw. Pooling k×n scores into one list is precisely what made
213
+ // generation noise read as judge noise: it is the defect Report #006 exists
214
+ // to name, and the nesting is the whole measurement.
215
+ const draws = [];
216
+ let last = null; // last MEASURED draw — carries the v0.4-shaped fields
217
+ let lastTranscript = null;
218
+ let action = { stop: false, reason: 'below_min' };
219
+ let fatal = null;
220
+
221
+ while (!action.stop && draws.length < SAMPLING.max) {
222
+ const drawIndex = draws.length;
223
+ try {
224
+ onProgress({ case: c.id, mode, phase: 'generate', draw: drawIndex });
225
+ const gen = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs: ctGen });
226
+ calls += 1;
227
+ if (budget) budget.add((gen.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
228
+ const generationHash = sha256(String(gen.text || ''));
229
+ onProgress({ case: c.id, mode, phase: 'judge', samples, draw: drawIndex });
230
+ const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ctJudge, samples });
231
+ calls += samples;
232
+ if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judgeModel, 'judge'));
233
+ const draw = {
234
+ draw_index: drawIndex,
235
+ generation_hash: generationHash,
236
+ status: 'measured',
237
+ samples: jr.caseResult.samples,
238
+ judge_sample_hashes: jr.caseResult.judge_sample_hashes,
239
+ mean: jr.caseResult.mean,
240
+ stddev: jr.caseResult.stddev,
241
+ };
242
+ if (hasUsage(gen.usage)) draw.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
243
+ if (jr.caseResult.judge_usage) draw.judge_usage = jr.caseResult.judge_usage;
244
+ draws.push(draw);
245
+ last = jr;
246
+ if (keepTranscripts) lastTranscript = { id: c.id, mode, generation: String(gen.text || ''), judge_outputs: jr.sampleTexts };
247
+ } catch (e) {
248
+ if (e && e.code === 'BUDGET_HARDSTOP') throw e; // budget hard-stop stays fatal
249
+ if (!isTimeout(e)) { fatal = e; break; } // non-timeout errors stay fatal
250
+ if (budget) {
251
+ try {
252
+ if (e.phase === 'judge') budget.add((e.judgeAttempts || 1) * perCallCostUSD(judgeModel, 'judge'));
253
+ else budget.add((e.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
254
+ } catch (be) { if (be && be.code === 'BUDGET_HARDSTOP') throw be; }
255
+ }
256
+ // F-009-L: the draw is UNMEASURED. No score, no fabricated samples, and
257
+ // it is excluded from every statistic rather than counted as a zero — a
258
+ // zero asserts a measurement, and a timeout is the absence of one.
259
+ draws.push({
260
+ draw_index: drawIndex,
261
+ generation_hash: null,
262
+ status: 'unmeasured',
263
+ reason: String((e && e.message) || 'timeout').slice(0, 200),
264
+ samples: [],
265
+ mean: null,
266
+ stddev: null,
267
+ });
268
+ onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: String((e && e.message) || 'timeout') });
198
269
  }
270
+ action = nextAction(draws);
271
+ }
272
+ if (fatal) throw fatal;
273
+
274
+ const agg = acrossDraws(draws);
275
+ const generation = {
276
+ n_planned: SAMPLING.min,
277
+ n_drawn: agg.n_drawn,
278
+ n_measured: agg.n_measured,
279
+ n_unmeasured: agg.n_unmeasured,
280
+ stopping_reason: action.reason,
281
+ mean: agg.mean,
282
+ sd: agg.sd,
283
+ judge_sd_mean: agg.judge_sd_mean,
284
+ variance_ratio: agg.variance_ratio,
285
+ // WHICH null, when it is null (F-014-C). Copied through explicitly rather
286
+ // than spread from `agg`: this assembly names its keys one by one, and the
287
+ // canary was dropped by exactly such an assembly silently gaining a field
288
+ // upstream that nothing here carried down (F-014-D).
289
+ variance_ratio_unavailable: agg.variance_ratio_unavailable,
290
+ draws,
291
+ };
292
+
293
+ // Every draw failed: the case is recorded failed_timeout as before, but it
294
+ // now carries the draw list showing WHAT failed and how often.
295
+ if (!last) {
199
296
  failedCases += 1;
200
- onProgress({ case: c.id, mode, phase: 'failed', reason: String((e && e.message) || 'timeout') });
201
- return { caseResult: { id: c.id, mode, case_status: 'failed_timeout', reason: String((e && e.message) || 'timeout').slice(0, 200) }, transcript: null };
297
+ return { caseResult: { id: c.id, mode, case_status: 'failed_timeout', reason: (draws[draws.length - 1] || {}).reason || 'timeout', generation }, transcript: null };
202
298
  }
299
+
300
+ // The v0.4-shaped fields now describe the DRAW SET, not one arbitrary draw,
301
+ // so a v0.4 reader pointed at a v0.5 receipt reads the aggregate rather than
302
+ // whichever draw happened to be last. `generation_hash`, `samples` and
303
+ // `judge_sample_hashes` continue to describe the last measured draw, which
304
+ // is the one they have always described; RECEIPT.md states this.
305
+ const caseResult = { ...last.caseResult, generation };
306
+ caseResult.mean = agg.mean;
307
+ caseResult.score = agg.mean;
308
+ caseResult.stddev = agg.sd;
309
+ caseResult.outcome = outcomeFor(agg.mean, agg.sd, c.pass_threshold);
310
+ onProgress({ case: c.id, mode, phase: 'done', outcome: caseResult.outcome, score: caseResult.mean, stddev: caseResult.stddev });
311
+ return { caseResult, transcript: lastTranscript };
203
312
  });
204
313
  const caseResults = pairs.map((p) => p.caseResult);
205
314
  const transcripts = keepTranscripts ? pairs.map((p) => p.transcript) : null;
@@ -229,7 +338,18 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
229
338
  // v0.3.1 value-per-token axis: estimated SKILL.md token size.
230
339
  tokens: estimateTokens(skill.skillMd),
231
340
  },
232
- suite: { format: skill.suite.format, suiteHash: skill.suite.suiteHash, caseCount: skill.suite.caseCount },
341
+ // v0.5: the suite canary. Derived from the suite identity and its case ids,
342
+ // so it is stable without a registry and distinct across suites — a leaked
343
+ // suite is detectable in a corpus. A detection aid, not a control.
344
+ suite: {
345
+ format: skill.suite.format,
346
+ suiteHash: skill.suite.suiteHash,
347
+ caseCount: skill.suite.caseCount,
348
+ // The FULL suite, never the post---max-cases list: a canary derived from a
349
+ // truncated run is not stable for the suite, and a canary that moves
350
+ // cannot say which suite leaked. Approval finding, spec 014.
351
+ canary: suiteCanary({ id: skill.name || skill.suite.suiteHash, cases: (skill.suite.cases || cases).map((c) => ({ id: c.id })) }),
352
+ },
233
353
  run: {
234
354
  model_id: modelId,
235
355
  model_release_date: releaseDateFor(modelId),
@@ -277,7 +397,7 @@ function summarizeReceipt(receipt) {
277
397
  const aggs = receipt.results.aggregates;
278
398
  const sign = cmp.delta >= 0 ? '+' : '';
279
399
  if (receipt.run.status === 'incomplete') {
280
- L.push(`> ⏱ **INCOMPLETE** — ${receipt.run.failed_case_count} case(s) failed (timed out after retries) and are EXCLUDED from the aggregates below; this receipt must not be used to compute a drift/durability verdict.`);
400
+ L.push(`> ⏱ **INCOMPLETE** — ${receipt.run.failed_case_count} case(s) had an arm that could not be measured and are EXCLUDED from the aggregates below, BOTH arms together; this receipt must not be used to compute a drift/durability verdict.`);
281
401
  L.push('');
282
402
  }
283
403
  L.push(`with_skill **${band(aggs.with_skill.mean_score, aggs.with_skill.stddev)}** vs baseline **${band(aggs.baseline.mean_score, aggs.baseline.stddev)}**`);
@@ -300,4 +420,4 @@ function summarizeReceipt(receipt) {
300
420
  return L.join('\n');
301
421
  }
302
422
 
303
- module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor };
423
+ module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor, resolveCallTimeoutMs };