driftproof 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/run.js CHANGED
@@ -2,16 +2,18 @@
2
2
  'use strict';
3
3
 
4
4
  const { complete, resolveModel, surfaceForModel, isMeteredSurface, retryPolicyForSurface, CODEX_OVERHEAD_NOTE } = require('./provider');
5
- const { gradeSamples, judgeSettings } = require('./judge');
6
- const { buildReceipt } = require('./receipt');
5
+ const { gradeSamples, judgeSettings, promptTemplateHash, isTruncated } = require('./judge');
6
+ const { buildReceipt, caseFailed, FAILED_STATUSES } = require('./receipt');
7
7
  const { sha256 } = require('./canonical');
8
- const { registryStatus, providerForModel, priceForModel } = require('./models');
8
+ const { registryStatus, providerForModel, priceForModel, assertRegistered } = require('./models');
9
9
  const { perCallCostUSD } = require('./cost');
10
10
  const { runChecks } = require('./checks');
11
11
  const { estimateTokens } = require('./skillCost');
12
12
  const { hasUsage, normalizeUsage } = require('./usage');
13
13
  const { buildPricingSnapshot, computeEconomics } = require('./value');
14
14
  const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES, DEV_MAX_CALLS } = require('../config');
15
+ const { stubEnabled } = require('./stub');
16
+ const { canonicalModelId } = require('./usage');
15
17
  const { SAMPLING, acrossDraws, nextAction } = require('./sampling');
16
18
  const { suiteCanary } = require('./canary');
17
19
 
@@ -51,7 +53,7 @@ function projectCalls(caseCount, samples, draws = 1) {
51
53
  // Ask the target model to perform one eval case. `withSkill` decides whether the
52
54
  // SKILL.md is prepended as a system prompt (the whole point: measure the skill's
53
55
  // marginal effect vs a bare baseline).
54
- async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs }) {
56
+ async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs, trusted = false }) {
55
57
  // Test seam (gate only): force a persistent timeout for a named case id so the
56
58
  // failed_timeout path is exercised deterministically without any live call.
57
59
  if (process.env.DRIFTPROOF_TEST_TIMEOUT_CASEID && process.env.DRIFTPROOF_TEST_TIMEOUT_CASEID === caseObj.id) {
@@ -60,8 +62,44 @@ async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs }) {
60
62
  throw e;
61
63
  }
62
64
  const system = withSkill ? skillMd : undefined;
63
- const { text, usage, wall_ms, attempts } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
64
- return { text, usage, wall_ms, attempts };
65
+ const out = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs, trusted });
66
+ // The whole reply travels: what answered, what it said served the call, why
67
+ // it stopped, and which spawn path was taken (spec 026 AC-1, AC-2, AC-8).
68
+ return { text: out.text, usage: out.usage, wall_ms: out.wall_ms, attempts: out.attempts, answeredBy: out.answeredBy, surface: out.surface, reportedModels: out.reportedModels, stopReason: out.stopReason, isolation: out.isolation };
69
+ }
70
+
71
+ // ── what answered (spec 026, AC-2) ───────────────────────────────────────────
72
+ // The surface's echo is compared with the requested id on CANONICAL ids (the
73
+ // real claude CLI keys its usage by the undated form). A surface that names a
74
+ // DIFFERENT model stops the run before the next call; no receipt is written,
75
+ // because a receipt naming a model that did not answer is the defect this
76
+ // exists to close. A surface that names nothing is recorded as attested: false.
77
+ function attest(reply, requestedId, phase) {
78
+ const reported = Array.isArray(reply && reply.reportedModels) ? reply.reportedModels : null;
79
+ if (!reported || !reported.length) return { attested: false, reported };
80
+ const want = canonicalModelId(resolveModel(requestedId));
81
+ const other = reported.find((id) => canonicalModelId(id) !== want);
82
+ if (other) {
83
+ const e = new Error(`substrate mismatch: the ${phase} surface answered as "${other}" where "${requestedId}" (canonical "${want}") was requested; the run stops before the next call and no receipt is written`);
84
+ e.code = 'SUBSTRATE_MISMATCH';
85
+ e.requested = requestedId; e.reported = other;
86
+ throw e;
87
+ }
88
+ return { attested: true, reported };
89
+ }
90
+
91
+ // The run's answered_by block, derived from every reply the run received and
92
+ // never from the surface the runner would have chosen (spec 026 AC-1: a stub
93
+ // run used to record the real surface name because the name was computed
94
+ // from the model id, not from what answered). With no reply at all (every
95
+ // call failed before answering) the kind is what the process would have
96
+ // answered with, which is the one thing still known.
97
+ function answeredByOf(replies, { attestedGen = false, reportedAll = new Set() } = {}) {
98
+ const kinds = new Set(replies.map((r) => r && r.answeredBy).filter(Boolean));
99
+ const kind = replies.length ? (kinds.size === 1 && kinds.has('stub') ? 'stub' : 'model') : (stubEnabled() ? 'stub' : 'model');
100
+ const iso = replies.map((r) => r && r.isolation).find(Boolean) || 'none';
101
+ const reportedModels = reportedAll.size ? [...reportedAll].sort() : null;
102
+ return { kind, attested: kind === 'model' && attestedGen, reported_models: reportedModels, isolation: iso };
65
103
  }
66
104
 
67
105
  // Determine a case outcome from its sampled band and threshold.
@@ -78,8 +116,14 @@ function outcomeFor(mean, stddev, threshold) {
78
116
  // `generationHash` binds the graded case to the exact generation text (v0.3).
79
117
  // Returns { caseResult, sampleTexts } — sampleTexts is transient (retained only
80
118
  // under --keep-transcripts; never part of the receipt).
81
- async function judgeCase({ caseObj, response, generationHash, judgeModel, mode, timeoutMs, samples }) {
82
- const g = await gradeSamples({ task: caseObj.prompt, response, rubric: caseObj.rubric, model: judgeModel, samples, timeoutMs });
119
+ async function judgeCase({ caseObj, response, generationHash, judgeModel, mode, timeoutMs, samples, trusted = false }) {
120
+ const g = await gradeSamples({ task: caseObj.prompt, response, rubric: caseObj.rubric, model: judgeModel, samples, timeoutMs, trusted });
121
+ // A judge that produced no score produced no measurement (spec 026 AC-3):
122
+ // the draw is unmeasured with the judge's reason, and no sample is kept.
123
+ // The judge calls that WERE made travel with the unmeasured result: their
124
+ // output hashes (transcript auditability: the judge said something, and a
125
+ // reader can check what) and their usage, so the receipt records every call.
126
+ if (g.unmeasured) return { unmeasured: true, reason: g.reason, sampleTexts: g.sample_texts, sampleHashes: g.sample_hashes || [], attempts: g.attempts, replies: g.replies || [], judge_usage: hasUsage(g.usage) ? g.usage : null };
83
127
  const outcome = outcomeFor(g.mean, g.stddev, caseObj.pass_threshold);
84
128
  const caseResult = {
85
129
  id: caseObj.id,
@@ -102,7 +146,7 @@ async function judgeCase({ caseObj, response, generationHash, judgeModel, mode,
102
146
  if (checks.length) caseResult.checks = checks;
103
147
  // v0.4: grading overhead for this case row, kept OUT of the skill-value math.
104
148
  if (hasUsage(g.usage)) caseResult.judge_usage = g.usage;
105
- return { caseResult, sampleTexts: g.sample_texts, attempts: g.attempts };
149
+ return { caseResult, sampleTexts: g.sample_texts, attempts: g.attempts, replies: g.replies || [] };
106
150
  }
107
151
 
108
152
  // Run up to `concurrency` async tasks at a time, preserving input order in the
@@ -127,7 +171,7 @@ async function mapPool(items, concurrency, fn) {
127
171
  // a sealed receipt. Enforces a hard call cap; every model+judge call counts.
128
172
  //
129
173
  // opts: { maxCases, maxCalls, samples, judgeModel, timeoutMs, concurrency,
130
- // onProgress, budget, keepTranscripts, nowIso }
174
+ // onProgress, budget, keepTranscripts, nowIso, trusted }
131
175
  // budget — optional BudgetTracker; accumulates estimated per-call
132
176
  // USD as the run proceeds and hard-stops at 1.25× the cap.
133
177
  // keepTranscripts — when true, the run records transcripts:"retained-local"
@@ -155,6 +199,13 @@ const resolveCallTimeoutMs = function resolveCallTimeoutMs(surface, opts = {}) {
155
199
  async function runSkillOnModel({ skill, model, opts = {} }) {
156
200
  const modelId = resolveModel(model);
157
201
  const judgeModel = opts.judgeModel ? resolveModel(opts.judgeModel) : modelId;
202
+ // Spec 026 AC-10 (F6): the target and the judge must be models the registry
203
+ // knows, checked HERE, on the path the CLI, the report scripts and the
204
+ // trigger all share, before the cost guard and before any call. bin/driftproof
205
+ // makes the same check at its door so the refusal names the registry path in
206
+ // its own message; this one is the door every caller passes.
207
+ assertRegistered(modelId, 'model');
208
+ assertRegistered(judgeModel, 'judge model');
158
209
  // THE SURFACE'S OWN DECLARED TIMEOUT, resolved per arm (spec 017 AC-1).
159
210
  //
160
211
  // This read `opts.timeoutMs || 120000`, and `lib/provider.js` documents that an
@@ -177,6 +228,10 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
177
228
  const onProgress = opts.onProgress || (() => {});
178
229
  const budget = opts.budget || null;
179
230
  const keepTranscripts = !!opts.keepTranscripts;
231
+ // spec 022: the SAME-USER legacy spawn is reachable only when a caller says
232
+ // `trusted: true` (bin/driftproof --trusted-skill). Every other caller of this
233
+ // function, the report scripts and the release watcher included, isolates.
234
+ const trusted = !!opts.trusted;
180
235
 
181
236
  let cases = skill.suite.cases;
182
237
  if (opts.maxCases && cases.length > opts.maxCases) cases = cases.slice(0, opts.maxCases);
@@ -200,6 +255,12 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
200
255
 
201
256
  let calls = 0;
202
257
  let failedCases = 0;
258
+ // Spec 026 AC-1, AC-2: every reply the run received, for the answered_by
259
+ // block; whether every generation reply attested the requested model (read
260
+ // from the surface's echo, AC-2); every canonical id any reply named.
261
+ const replies = [];
262
+ let attestedGen = true;
263
+ const reportedAll = new Set();
203
264
  const isTimeout = (e) => !!(e && (e.code === 'TIMEOUT' || /tim(e|ed)\s*out/i.test(String((e && e.message) || ''))));
204
265
  const genKind = (ws) => (ws ? 'gen_with_skill' : 'gen_baseline');
205
266
  const pairs = await mapPool(tasks, concurrency, async ({ c, withSkill }) => {
@@ -217,19 +278,93 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
217
278
  let lastTranscript = null;
218
279
  let action = { stop: false, reason: 'below_min' };
219
280
  let fatal = null;
281
+ // Spec 026 AC-3: an unmeasured draw that was NOT a timeout (an empty
282
+ // generation, a judge with no score) makes the case failed_unmeasured
283
+ // rather than failed_timeout when no draw measured.
284
+ let nonTimeout = false;
285
+
286
+ // An UNMEASURED draw carries no score, no fabricated samples, and the
287
+ // reason it carries none; it is excluded from every statistic rather than
288
+ // counted as a zero. v0.6 adds what the surface said about the reply. The
289
+ // generation's own hash is kept when there was text.
290
+ const unmeasuredDraw = (drawIndex, reason, gen) => ({
291
+ draw_index: drawIndex,
292
+ generation_hash: gen && String(gen.text || '') ? sha256(String(gen.text || '')) : null,
293
+ status: 'unmeasured',
294
+ reason: String(reason || '').slice(0, 200),
295
+ samples: [],
296
+ mean: null,
297
+ stddev: null,
298
+ stop_reason: gen ? (gen.stopReason || null) : null,
299
+ truncated: !!(gen && gen.truncated === true),
300
+ reported_model: gen && gen.attested ? modelId : null,
301
+ });
220
302
 
221
303
  while (!action.stop && draws.length < SAMPLING.max) {
222
304
  const drawIndex = draws.length;
223
305
  try {
224
306
  onProgress({ case: c.id, mode, phase: 'generate', draw: drawIndex });
225
- const gen = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs: ctGen });
307
+ const gen = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs: ctGen, trusted });
226
308
  calls += 1;
227
309
  if (budget) budget.add((gen.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
310
+ replies.push(gen);
311
+ // What answered, on canonical ids; a different model stops the run.
312
+ const a = attest(gen, modelId, 'generation');
313
+ gen.attested = a.attested;
314
+ if (!a.attested) attestedGen = false;
315
+ for (const id of (a.reported || [])) reportedAll.add(canonicalModelId(id));
316
+ gen.truncated = isTruncated(gen.stopReason);
317
+ // Spec 026 AC-8 (F4): a generation cut at the output cap is a partial
318
+ // answer, and a partial answer graded as a whole one is a score about
319
+ // something the model did not write. Recorded truncated, unmeasured,
320
+ // never judged.
321
+ if (gen.truncated === true) {
322
+ nonTimeout = true;
323
+ const d = unmeasuredDraw(drawIndex, `generation truncated at the output cap (stop_reason ${gen.stopReason})`, gen);
324
+ if (hasUsage(gen.usage)) d.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
325
+ draws.push(d);
326
+ onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: d.reason });
327
+ action = nextAction(draws);
328
+ continue;
329
+ }
330
+ // Spec 026 AC-3 (F2): an empty generation is the absence of an output,
331
+ // not an output that scored zero. It is not sent to the judge.
332
+ if (!String(gen.text || '').trim()) {
333
+ nonTimeout = true;
334
+ const d = unmeasuredDraw(drawIndex, 'empty generation (the surface returned no text)', gen);
335
+ if (hasUsage(gen.usage)) d.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
336
+ draws.push(d);
337
+ onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: d.reason });
338
+ action = nextAction(draws);
339
+ continue;
340
+ }
228
341
  const generationHash = sha256(String(gen.text || ''));
229
342
  onProgress({ case: c.id, mode, phase: 'judge', samples, draw: drawIndex });
230
- const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ctJudge, samples });
343
+ const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ctJudge, samples, trusted });
231
344
  calls += samples;
232
345
  if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judgeModel, 'judge'));
346
+ for (const r of (jr.replies || [])) {
347
+ if (!r) continue;
348
+ replies.push(r);
349
+ const ja = attest(r, judgeModel, 'judge');
350
+ for (const id of (ja.reported || [])) reportedAll.add(canonicalModelId(id));
351
+ }
352
+ if (jr.unmeasured) {
353
+ // The judge returned no score for this draw (empty, unparseable,
354
+ // non-numeric, out of range): unmeasured, naming the judge's reason;
355
+ // the generation's own hash is kept.
356
+ nonTimeout = true;
357
+ const d = unmeasuredDraw(drawIndex, jr.reason, gen);
358
+ // The judge samples taken before the draw was called unmeasured are
359
+ // recorded by hash (no score entered any statistic; the calls happened).
360
+ if ((jr.sampleHashes || []).length) d.judge_sample_hashes = jr.sampleHashes;
361
+ if (hasUsage(gen.usage)) d.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
362
+ if (jr.judge_usage) d.judge_usage = jr.judge_usage;
363
+ draws.push(d);
364
+ onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: d.reason });
365
+ action = nextAction(draws);
366
+ continue;
367
+ }
233
368
  const draw = {
234
369
  draw_index: drawIndex,
235
370
  generation_hash: generationHash,
@@ -238,6 +373,12 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
238
373
  judge_sample_hashes: jr.caseResult.judge_sample_hashes,
239
374
  mean: jr.caseResult.mean,
240
375
  stddev: jr.caseResult.stddev,
376
+ // v0.6 (spec 026 AC-2, AC-8): why the generation stopped, that it
377
+ // was not cut (a cut draw never reaches here), and what the surface
378
+ // said served it (null when it said nothing).
379
+ stop_reason: gen.stopReason || null,
380
+ truncated: false,
381
+ reported_model: gen.attested ? modelId : null,
241
382
  };
242
383
  if (hasUsage(gen.usage)) draw.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
243
384
  if (jr.caseResult.judge_usage) draw.judge_usage = jr.caseResult.judge_usage;
@@ -246,7 +387,7 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
246
387
  if (keepTranscripts) lastTranscript = { id: c.id, mode, generation: String(gen.text || ''), judge_outputs: jr.sampleTexts };
247
388
  } catch (e) {
248
389
  if (e && e.code === 'BUDGET_HARDSTOP') throw e; // budget hard-stop stays fatal
249
- if (!isTimeout(e)) { fatal = e; break; } // non-timeout errors stay fatal
390
+ if (!isTimeout(e)) { fatal = e; break; } // non-timeout errors stay fatal (a SUBSTRATE_MISMATCH among them)
250
391
  if (budget) {
251
392
  try {
252
393
  if (e.phase === 'judge') budget.add((e.judgeAttempts || 1) * perCallCostUSD(judgeModel, 'judge'));
@@ -264,6 +405,9 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
264
405
  samples: [],
265
406
  mean: null,
266
407
  stddev: null,
408
+ stop_reason: null,
409
+ truncated: false,
410
+ reported_model: null,
267
411
  });
268
412
  onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: String((e && e.message) || 'timeout') });
269
413
  }
@@ -287,14 +431,22 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
287
431
  // canary was dropped by exactly such an assembly silently gaining a field
288
432
  // upstream that nothing here carried down (F-014-D).
289
433
  variance_ratio_unavailable: agg.variance_ratio_unavailable,
434
+ // v0.6 (spec 026 AC-8): draws cut at the output cap, counted here so a
435
+ // reader sees it without walking the draw list.
436
+ n_truncated: draws.filter((d) => d.truncated === true).length,
290
437
  draws,
291
438
  };
292
439
 
293
- // Every draw failed: the case is recorded failed_timeout as before, but it
294
- // now carries the draw list showing WHAT failed and how often.
440
+ // Every draw failed: the case is recorded failed, carrying the draw list
441
+ // showing WHAT failed and how often. failed_timeout when every failure was
442
+ // a timeout; failed_unmeasured when any draw was unmeasured for another
443
+ // reason (spec 026 AC-3). The two share one predicate, caseFailed, in
444
+ // lib/receipt.js, and no reader excludes by either literal.
295
445
  if (!last) {
296
446
  failedCases += 1;
297
- return { caseResult: { id: c.id, mode, case_status: 'failed_timeout', reason: (draws[draws.length - 1] || {}).reason || 'timeout', generation }, transcript: null };
447
+ const status = nonTimeout ? FAILED_STATUSES[1] : FAILED_STATUSES[0];
448
+ const lastReason = [...draws].reverse().map((d) => d.reason).find(Boolean) || 'timeout';
449
+ return { caseResult: { id: c.id, mode, case_status: status, reason: lastReason, generation }, transcript: null };
298
450
  }
299
451
 
300
452
  // The v0.4-shaped fields now describe the DRAW SET, not one arbitrary draw,
@@ -313,7 +465,14 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
313
465
  const caseResults = pairs.map((p) => p.caseResult);
314
466
  const transcripts = keepTranscripts ? pairs.map((p) => p.transcript) : null;
315
467
 
316
- const surface = surfaceForModel(modelId);
468
+ // THE SURFACE IS WHAT ANSWERED, not what the runner would have chosen for
469
+ // the model id (spec 026 AC-1, F1). A stub run records surface stub, judge
470
+ // surface stub, answered_by.kind stub, and is UNVERIFIED: it measured nothing.
471
+ const answered = answeredByOf(replies, { attestedGen: attestedGen && replies.some((r) => r && r.answeredBy === 'model'), reportedAll });
472
+ const answeredBy = { kind: answered.kind, attested: answered.attested, reported_model: answered.attested ? modelId : null, reported_models: answered.reported_models, isolation: answered.isolation };
473
+ const surface = answered.kind === 'stub' ? 'stub' : surfaceForModel(modelId);
474
+ // v0.6 (spec 026 AC-11): which judge ran, and the template it graded with.
475
+ const judgeBlock = { ...judgeSettings(samples, judgeModel), ...(answered.kind === 'stub' ? { surface: 'stub' } : {}), model_id: judgeModel, prompt_template_hash: promptTemplateHash() };
317
476
  const nowIso = opts.nowIso || new Date().toISOString();
318
477
  // v0.4 economics. The pricing snapshot is frozen HERE, at run time, from the
319
478
  // registry; every derived dollar figure below is computed from the snapshot and
@@ -355,24 +514,52 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
355
514
  model_release_date: releaseDateFor(modelId),
356
515
  provider: providerForModel(modelId),
357
516
  surface,
358
- // v0.3.1: on the openai/cli (codex) surface, record the fixed harness preamble.
517
+ // v0.3.1: on the openai/cli (codex) surface, record the fixed harness
518
+ // preamble. Absent on a stub run: it describes a harness that did not run.
359
519
  surface_overhead_note: surface === 'openai-cli' ? CODEX_OVERHEAD_NOTE : undefined,
360
520
  runner_version: RUNNER_VERSION,
361
521
  date_utc: nowIso,
362
522
  registry: registryStatus(modelId),
363
523
  transcripts: keepTranscripts ? 'retained-local' : 'hashes-only',
364
- judge: judgeSettings(samples, judgeModel),
524
+ judge: judgeBlock,
365
525
  pricing_snapshot: pricingSnapshot,
526
+ answered_by: answeredBy,
366
527
  },
367
528
  cases: caseResults,
368
529
  economics,
369
- verificationLevel: 'TESTED',
530
+ // The level is DERIVED from what answered: only a model-answered run may
531
+ // read TESTED. The schema refuses TESTED on a stub receipt as the second,
532
+ // independent control (spec 026 AC-1).
533
+ verificationLevel: answered.kind === 'model' ? 'TESTED' : 'UNVERIFIED',
370
534
  });
371
535
 
372
536
  return { receipt, calls, transcripts, failedCases };
373
537
  }
374
538
 
375
- function band(mean, sd) { return `${mean.toFixed(3)} ± ${sd.toFixed(3)}`; }
539
+ // A band the formula could not form is printed as what it is, never as 0.000
540
+ // (spec 026 AC-7): one included case has a mean and no dispersion; none has
541
+ // neither.
542
+ function band(mean, sd) {
543
+ if (mean == null) return 'n/a (0 cases)';
544
+ if (sd == null) return `${mean.toFixed(3)} ± n/a (1 case)`;
545
+ return `${mean.toFixed(3)} ± ${sd.toFixed(3)}`;
546
+ }
547
+ // The comparison band, or the reason there is none.
548
+ function uncertaintyStr(cmp) {
549
+ if (cmp.delta_uncertainty != null) return cmp.delta_uncertainty.toFixed(3);
550
+ return `n/a (${cmp.delta_uncertainty_unavailable === 'single_case' ? '1 case' : cmp.delta_uncertainty_unavailable === 'no_cases' ? '0 cases' : 'no band'})`;
551
+ }
552
+ // The answered_by block, said in one line for the summary and the CLI.
553
+ function answeredLine(receipt) {
554
+ const ab = (receipt.run && receipt.run.answered_by) || null;
555
+ if (!ab) return 'answered by: unrecorded (pre-v0.6 receipt)';
556
+ if (ab.kind === 'stub') return 'answered by: stub (DRIFTPROOF_STUB) — nothing answered; this run measured nothing (UNVERIFIED)';
557
+ if (ab.kind === 'external') return 'answered by: an external tool (imported)';
558
+ const iso = ab.isolation === 'eval-user' ? 'the isolated eval-user hop' : ab.isolation === 'same-user' ? 'the same-user spawn (--trusted-skill)' : 'no spawn (api surface)';
559
+ return ab.attested
560
+ ? `answered by: model ${ab.reported_model} (attested by the surface; ${iso})`
561
+ : `answered by: model, but the surface did not report which model answered (attested: false; ${iso})`;
562
+ }
376
563
 
377
564
  // Render a short human-readable markdown summary of a receipt.
378
565
  function summarizeReceipt(receipt) {
@@ -381,6 +568,7 @@ function summarizeReceipt(receipt) {
381
568
  L.push('');
382
569
  L.push(`- **model:** \`${receipt.run.model_id}\`${receipt.run.model_release_date ? ` (released ${receipt.run.model_release_date})` : ''}`);
383
570
  L.push(`- **surface:** ${receipt.run.surface}`);
571
+ L.push(`- **${answeredLine(receipt)}**`);
384
572
  L.push(`- **run (UTC):** ${receipt.run.date_utc}`);
385
573
  L.push(`- **runner:** v${receipt.run.runner_version}`);
386
574
  const j = receipt.run.judge || {};
@@ -395,22 +583,39 @@ function summarizeReceipt(receipt) {
395
583
  L.push('');
396
584
  const cmp = receipt.comparison;
397
585
  const aggs = receipt.results.aggregates;
398
- const sign = cmp.delta >= 0 ? '+' : '';
586
+ if ((receipt.run.answered_by || {}).kind === 'stub') {
587
+ L.push('> **STUB RUN** — DRIFTPROOF_STUB=1: nothing answered, the text was canned, and this run measured nothing. The receipt is UNVERIFIED and verdicts nothing.');
588
+ L.push('');
589
+ }
399
590
  if (receipt.run.status === 'incomplete') {
400
591
  L.push(`> ⏱ **INCOMPLETE** — ${receipt.run.failed_case_count} case(s) had an arm that could not be measured and are EXCLUDED from the aggregates below, BOTH arms together; this receipt must not be used to compute a drift/durability verdict.`);
401
592
  L.push('');
402
593
  }
594
+ // Spec 026 AC-8: draws cut at the output cap, said once for the run.
595
+ const nTruncated = receipt.results.cases.reduce((a, c) => a + (((c.generation || {}).n_truncated) || 0), 0);
596
+ if (nTruncated) {
597
+ L.push(`> ✂ **${nTruncated} draw(s) truncated** at the output cap: each is unmeasured and excluded from every band below.`);
598
+ L.push('');
599
+ }
403
600
  L.push(`with_skill **${band(aggs.with_skill.mean_score, aggs.with_skill.stddev)}** vs baseline **${band(aggs.baseline.mean_score, aggs.baseline.stddev)}**`);
404
601
  L.push('');
405
- L.push(`skill lift **${sign}${cmp.delta.toFixed(3)}** (combined uncertainty ± ${cmp.delta_uncertainty.toFixed(3)})`);
602
+ if (cmp.delta == null) {
603
+ L.push(`skill lift **n/a** (${cmp.delta_uncertainty_unavailable === 'no_cases' ? 'no case was included on an arm' : 'no comparison'})`);
604
+ } else {
605
+ const sign = cmp.delta >= 0 ? '+' : '';
606
+ L.push(`skill lift **${sign}${cmp.delta.toFixed(3)}** (combined uncertainty ± ${uncertaintyStr(cmp)})`);
607
+ }
608
+ L.push('');
609
+ // The rule the bands above are derived by, said beside them (spec 026 AC-6).
610
+ L.push(`band rule: each arm's band is the sample stddev of its per-case means${aggs.band_rule ? ` — ${aggs.band_rule}` : ' (unstated on this pre-v0.6 receipt)'}`);
406
611
  L.push('');
407
612
  L.push(`## Per-case (mean ± stddev over ${(receipt.run.judge || {}).samples || 1} judge samples)`);
408
613
  L.push('');
409
614
  L.push(`| case | mode | outcome | mean ± stddev | judge reason |`);
410
615
  L.push(`|---|---|---|---|---|`);
411
616
  for (const c of receipt.results.cases) {
412
- if (c.case_status === 'failed_timeout') {
413
- L.push(`| \`${c.id}\` | ${c.mode} | ⏱ failed_timeout | — (not measured) | ${c.reason || 'timed out'} |`);
617
+ if (caseFailed(c)) {
618
+ L.push(`| \`${c.id}\` | ${c.mode} | ⏱ ${c.case_status} | — (not measured) | ${c.reason || 'not measured'} |`);
414
619
  continue;
415
620
  }
416
621
  const flag = c.outcome === 'borderline' ? ' ⚠' : '';
@@ -420,4 +625,4 @@ function summarizeReceipt(receipt) {
420
625
  return L.join('\n');
421
626
  }
422
627
 
423
- module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor, resolveCallTimeoutMs };
628
+ module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor, resolveCallTimeoutMs, answeredLine, answeredByOf, attest, band, uncertaintyStr };
package/lib/skill.js CHANGED
@@ -4,7 +4,16 @@
4
4
  const fs = require('fs');
5
5
  const path = require('path');
6
6
  const { sha256Files, sha256Canonical } = require('./canonical');
7
- const { SUITE_FORMAT } = require('../config');
7
+ const { SUITE_FORMAT, SKILL_MAX_FILES, SKILL_MAX_BYTES, SKILL_MAX_DEPTH, SUITE_MAX_CASES, CASE_MAX_CHARS } = require('../config');
8
+
9
+ // A bound was exceeded: the error names the bound and the value (spec 026
10
+ // AC-13), so the refusal says what to change rather than that something is
11
+ // too big.
12
+ function pastBound(bound, limit, value, what) {
13
+ const e = new Error(`${what}: ${value} exceeds ${bound} (${limit}); refusing to load`);
14
+ e.code = 'INPUT_BOUND'; e.bound = bound; e.limit = limit; e.value = value;
15
+ return e;
16
+ }
8
17
 
9
18
  // Load a skill directory and its eval suite.
10
19
  //
@@ -21,13 +30,22 @@ const { SUITE_FORMAT } = require('../config');
21
30
  const IGNORE_DIRS = new Set(['.git', 'node_modules', 'evals']);
22
31
  const IGNORE_FILES = new Set(['.DS_Store']);
23
32
 
24
- function walkFiles(dir, base = dir, acc = []) {
33
+ // Bounded (spec 026 AC-13): the walk stops at SKILL_MAX_DEPTH directories
34
+ // below the skill dir, SKILL_MAX_FILES bundled files, and SKILL_MAX_BYTES of
35
+ // them together, and throws naming the bound the moment one is passed, so a
36
+ // pathological tree is refused before its bytes are read into memory.
37
+ function walkFiles(dir, base = dir, acc = [], state = { bytes: 0 }, depth = 0) {
38
+ if (depth > SKILL_MAX_DEPTH) throw pastBound('SKILL_MAX_DEPTH', SKILL_MAX_DEPTH, depth, `directory depth under ${base}`);
25
39
  for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
26
40
  if (entry.isDirectory()) {
27
41
  if (IGNORE_DIRS.has(entry.name)) continue;
28
- walkFiles(path.join(dir, entry.name), base, acc);
42
+ walkFiles(path.join(dir, entry.name), base, acc, state, depth + 1);
29
43
  } else if (entry.isFile() && !IGNORE_FILES.has(entry.name)) {
30
44
  const abs = path.join(dir, entry.name);
45
+ if (acc.length + 1 > SKILL_MAX_FILES) throw pastBound('SKILL_MAX_FILES', SKILL_MAX_FILES, acc.length + 1, `bundled files under ${base}`);
46
+ const size = fs.statSync(abs).size;
47
+ state.bytes += size;
48
+ if (state.bytes > SKILL_MAX_BYTES) throw pastBound('SKILL_MAX_BYTES', SKILL_MAX_BYTES, state.bytes, `bundled bytes under ${base}`);
31
49
  acc.push({ path: path.relative(base, abs), bytes: fs.readFileSync(abs) });
32
50
  }
33
51
  }
@@ -108,12 +126,16 @@ function normalizeCases(raw) {
108
126
  : Array.isArray(raw.evals) ? raw.evals
109
127
  : null;
110
128
  if (!list) throw new Error('evals.json must be an array or have a `cases`/`evals` array');
129
+ // Bounded (spec 026 AC-13): the case count, and each prompt and rubric.
130
+ if (list.length > SUITE_MAX_CASES) throw pastBound('SUITE_MAX_CASES', SUITE_MAX_CASES, list.length, 'cases in the suite');
111
131
  return list.map((c, i) => {
112
132
  const id = String(c.id || c.name || `case-${i + 1}`);
113
133
  const prompt = c.prompt || c.input || c.task;
114
134
  const rubric = c.rubric || c.criteria || c.expected;
115
135
  if (!prompt) throw new Error(`case "${id}" is missing a prompt/input/task`);
116
136
  if (!rubric) throw new Error(`case "${id}" is missing a rubric/criteria/expected`);
137
+ if (String(prompt).length > CASE_MAX_CHARS) throw pastBound('CASE_MAX_CHARS', CASE_MAX_CHARS, String(prompt).length, `case "${id}" prompt characters`);
138
+ if (String(rubric).length > CASE_MAX_CHARS) throw pastBound('CASE_MAX_CHARS', CASE_MAX_CHARS, String(rubric).length, `case "${id}" rubric characters`);
117
139
  const threshold = typeof c.pass_threshold === 'number' ? c.pass_threshold
118
140
  : typeof c.threshold === 'number' ? c.threshold : 0.7;
119
141
  const norm = { id, prompt: String(prompt), rubric: String(rubric), pass_threshold: threshold };
@@ -124,4 +146,4 @@ function normalizeCases(raw) {
124
146
  });
125
147
  }
126
148
 
127
- module.exports = { loadSkill, normalizeCases, parseSkillMeta };
149
+ module.exports = { loadSkill, normalizeCases, parseSkillMeta, pastBound };
package/lib/stats.js CHANGED
@@ -31,8 +31,11 @@ function stderr(xs) {
31
31
  return round(stddev(xs) / Math.sqrt(n));
32
32
  }
33
33
 
34
- // Combine independent uncertainties in quadrature: sqrt(a^2 + b^2).
34
+ // Combine independent uncertainties in quadrature: sqrt(a^2 + b^2). Null when
35
+ // either band is null: a combination of a band that could not form cannot form
36
+ // either (spec 026 AC-7), and the receipt says why beside it.
35
37
  function combineUncertainty(a, b) {
38
+ if (a == null || b == null) return null;
36
39
  return round(Math.sqrt(a * a + b * b));
37
40
  }
38
41
 
@@ -45,10 +48,16 @@ function combineUncertainty(a, b) {
45
48
  // is a conventional, honest "mean ± stddev across the suite". The drift HEADLINE
46
49
  // verdict is driven by the per-case band-overlap verdicts (see lib/diff.js), not
47
50
  // by this aggregate band; this value is a reported summary statistic.
51
+ //
52
+ // A BAND THE FORMULA CANNOT FORM IS NULL, NEVER 0 (spec 026 AC-7, F3). The
53
+ // sample standard deviation of one value is undefined, and the mean of no
54
+ // values is not a number; printing 0.000 for either asserted a precision that
55
+ // was never measured (the one-case run's `± 0.000`). The rule the receipt
56
+ // states (results.aggregates.band_rule) is this function.
48
57
  function aggregateBands(cases) {
49
- if (!cases.length) return { mean: 0, stddev: 0 };
58
+ if (!cases.length) return { mean: null, stddev: null };
50
59
  const means = cases.map((c) => c.mean);
51
- return { mean: mean(means), stddev: stddev(means) };
60
+ return { mean: mean(means), stddev: means.length < 2 ? null : stddev(means) };
52
61
  }
53
62
 
54
63
  // Do two confidence bands (mean ± half-width) fail to overlap, and in which
package/lib/usage.js CHANGED
@@ -42,6 +42,27 @@
42
42
 
43
43
  function n(x) { return Number.isFinite(Number(x)) ? Number(x) : 0; }
44
44
 
45
+ // ── what answered (spec 026, AC-2 and AC-8) ──────────────────────────────────
46
+ // Every lane also reports, when it can, WHICH model served the call and WHY the
47
+ // reply stopped. Both are read here, never guessed: a surface that says nothing
48
+ // yields null for both, and the receipt records that it said nothing.
49
+ //
50
+ // reportedModels the ids the surface named, VERBATIM (the real claude CLI
51
+ // keys the call's usage under the undated form and puts a
52
+ // dated side entry beside it; an api response names one
53
+ // model). Null when the surface names no model. The runner
54
+ // compares on canonical ids (canonicalModelId: a trailing
55
+ // -YYYYMMDD is not part of the identity), never the reader.
56
+ // stopReason the surface's own word for why the reply ended (end_turn,
57
+ // max_tokens, stop, length ...). Null when it gives none.
58
+ function canonicalModelId(id) { return String(id || '').replace(/-\d{8}$/, ''); }
59
+ function reportedModelsOf(map) {
60
+ if (!map || typeof map !== 'object' || Array.isArray(map)) return null;
61
+ const ids = [...new Set(Object.keys(map).filter((k) => k))].sort();
62
+ return ids.length ? ids : null;
63
+ }
64
+ function stopReasonOf(v) { return typeof v === 'string' && v ? v : null; }
65
+
45
66
  // The empty/unknown usage record. Deliberately null (not zeros) so "the surface
46
67
  // did not tell us" never reads as "the call cost nothing".
47
68
  function emptyUsage() {
@@ -75,6 +96,9 @@ function parseClaudeCliJson(stdout) {
75
96
  text: typeof j.result === 'string' ? j.result : '',
76
97
  usage,
77
98
  isError: j.is_error === true,
99
+ // v0.6: the CLI's own stop reason and the models it says served the call.
100
+ stopReason: stopReasonOf(j.stop_reason),
101
+ reportedModels: reportedModelsOf(j.modelUsage),
78
102
  // The CLI reports its own dollar figure. Recorded here for completeness but
79
103
  // NOT used: costs are computed uniformly from the frozen pricing snapshot so
80
104
  // three substrates are on one basis (see lib/value.js).
@@ -85,7 +109,9 @@ function parseClaudeCliJson(stdout) {
85
109
  // ── openai/cli — `codex exec --json` JSONL event stream ───────────────────────
86
110
  // One JSON object per line. Usage rides the terminal `turn.completed` event;
87
111
  // input_tokens already INCLUDES cached_input_tokens. Unparseable lines are
88
- // skipped (the stream also carries progress events we do not model).
112
+ // skipped (the stream also carries progress events we do not model). The
113
+ // stream names neither the model that served the turn nor a stop reason, so
114
+ // both read null here and the receipt says the surface did not report them.
89
115
  function parseCodexJsonl(stdout) {
90
116
  const lines = String(stdout || '').split('\n').map((l) => l.trim()).filter(Boolean);
91
117
  let usage = null;
@@ -103,16 +129,30 @@ function parseCodexJsonl(stdout) {
103
129
  wall_ms: null,
104
130
  };
105
131
  }
106
- // The final message is normally read from the -o file; this is a fallback.
132
+ // The final message IS the stream: the last `item.completed` /
133
+ // `agent_message` event carries it. Nothing is read from a file after the
134
+ // call, on either spawn mode (spec 022 AC-9; the earlier comment here
135
+ // described an output file the lane no longer reads, F-022-3).
107
136
  if (ev.type === 'item.completed' && ev.item && ev.item.type === 'agent_message' && typeof ev.item.text === 'string') {
108
137
  text = ev.item.text;
109
138
  }
110
139
  }
111
- return { usage, text };
140
+ return { usage, text, stopReason: null, reportedModels: null };
112
141
  }
113
142
 
114
143
  // ── anthropic/api ─────────────────────────────────────────────────────────────
115
144
  // The Messages API reports input_tokens EXCLUDING cache, same as the CLI.
145
+ // The response object also names the model that served it (`model`) and why
146
+ // it stopped (`stop_reason`); read here from the object, null when absent.
147
+ // Exercised against tests/fixtures/response-anthropic-api.json, which is
148
+ // SYNTHESISED from the API reference, not captured from a call.
149
+ function readAnthropicApiResponse(resp) {
150
+ const r = resp && typeof resp === 'object' ? resp : {};
151
+ return {
152
+ stopReason: stopReasonOf(r.stop_reason),
153
+ reportedModels: typeof r.model === 'string' && r.model ? [r.model] : null,
154
+ };
155
+ }
116
156
  function parseAnthropicApiUsage(u) {
117
157
  if (!u) return emptyUsage();
118
158
  const cacheRead = n(u.cache_read_input_tokens);
@@ -127,7 +167,18 @@ function parseAnthropicApiUsage(u) {
127
167
 
128
168
  // ── openai/api (Chat Completions-compatible) ──────────────────────────────────
129
169
  // prompt_tokens is the total; the cached portion, when present, is nested under
130
- // prompt_tokens_details.cached_tokens.
170
+ // prompt_tokens_details.cached_tokens. The response also names the serving
171
+ // model (`model`) and the first choice's `finish_reason`; read here, null
172
+ // when absent. Exercised against tests/fixtures/response-openai-api.json,
173
+ // SYNTHESISED from the API reference, not captured from a call.
174
+ function readOpenaiApiResponse(resp) {
175
+ const r = resp && typeof resp === 'object' ? resp : {};
176
+ const choice = Array.isArray(r.choices) && r.choices[0] && typeof r.choices[0] === 'object' ? r.choices[0] : {};
177
+ return {
178
+ stopReason: stopReasonOf(choice.finish_reason),
179
+ reportedModels: typeof r.model === 'string' && r.model ? [r.model] : null,
180
+ };
181
+ }
131
182
  function parseOpenaiApiUsage(u) {
132
183
  if (!u) return emptyUsage();
133
184
  const details = u.prompt_tokens_details || {};
@@ -165,4 +216,5 @@ function normalizeUsage(u) {
165
216
  module.exports = {
166
217
  emptyUsage, hasUsage, sumUsage, normalizeUsage,
167
218
  parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage,
219
+ readAnthropicApiResponse, readOpenaiApiResponse, canonicalModelId, reportedModelsOf,
168
220
  };