driftproof 0.8.1 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/run.js CHANGED
@@ -2,16 +2,18 @@
2
2
  'use strict';
3
3
 
4
4
  const { complete, resolveModel, surfaceForModel, isMeteredSurface, retryPolicyForSurface, CODEX_OVERHEAD_NOTE } = require('./provider');
5
- const { gradeSamples, judgeSettings } = require('./judge');
6
- const { buildReceipt } = require('./receipt');
5
+ const { gradeSamples, judgeSettings, promptTemplateHash, isTruncated } = require('./judge');
6
+ const { buildReceipt, caseFailed, FAILED_STATUSES } = require('./receipt');
7
7
  const { sha256 } = require('./canonical');
8
- const { registryStatus, providerForModel, priceForModel } = require('./models');
8
+ const { registryStatus, providerForModel, priceForModel, assertRegistered } = require('./models');
9
9
  const { perCallCostUSD } = require('./cost');
10
10
  const { runChecks } = require('./checks');
11
11
  const { estimateTokens } = require('./skillCost');
12
12
  const { hasUsage, normalizeUsage } = require('./usage');
13
13
  const { buildPricingSnapshot, computeEconomics } = require('./value');
14
14
  const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES, DEV_MAX_CALLS } = require('../config');
15
+ const { stubEnabled } = require('./stub');
16
+ const { canonicalModelId } = require('./usage');
15
17
  const { SAMPLING, acrossDraws, nextAction } = require('./sampling');
16
18
  const { suiteCanary } = require('./canary');
17
19
 
@@ -60,8 +62,44 @@ async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs, trusted
60
62
  throw e;
61
63
  }
62
64
  const system = withSkill ? skillMd : undefined;
63
- const { text, usage, wall_ms, attempts } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs, trusted });
64
- return { text, usage, wall_ms, attempts };
65
+ const out = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs, trusted });
66
+ // The whole reply travels: what answered, what it said served the call, why
67
+ // it stopped, and which spawn path was taken (spec 026 AC-1, AC-2, AC-8).
68
+ return { text: out.text, usage: out.usage, wall_ms: out.wall_ms, attempts: out.attempts, answeredBy: out.answeredBy, surface: out.surface, reportedModels: out.reportedModels, stopReason: out.stopReason, isolation: out.isolation };
69
+ }
70
+
71
+ // ── what answered (spec 026, AC-2) ───────────────────────────────────────────
72
+ // The surface's echo is compared with the requested id on CANONICAL ids (the
73
+ // real claude CLI keys its usage by the undated form). A surface that names a
74
+ // DIFFERENT model stops the run before the next call; no receipt is written,
75
+ // because a receipt naming a model that did not answer is the defect this
76
+ // exists to close. A surface that names nothing is recorded as attested: false.
77
+ function attest(reply, requestedId, phase) {
78
+ const reported = Array.isArray(reply && reply.reportedModels) ? reply.reportedModels : null;
79
+ if (!reported || !reported.length) return { attested: false, reported };
80
+ const want = canonicalModelId(resolveModel(requestedId));
81
+ const other = reported.find((id) => canonicalModelId(id) !== want);
82
+ if (other) {
83
+ const e = new Error(`substrate mismatch: the ${phase} surface answered as "${other}" where "${requestedId}" (canonical "${want}") was requested; the run stops before the next call and no receipt is written`);
84
+ e.code = 'SUBSTRATE_MISMATCH';
85
+ e.requested = requestedId; e.reported = other;
86
+ throw e;
87
+ }
88
+ return { attested: true, reported };
89
+ }
90
+
91
+ // The run's answered_by block, derived from every reply the run received and
92
+ // never from the surface the runner would have chosen (spec 026 AC-1: a stub
93
+ // run used to record the real surface name because the name was computed
94
+ // from the model id, not from what answered). With no reply at all (every
95
+ // call failed before answering) the kind is what the process would have
96
+ // answered with, which is the one thing still known.
97
+ function answeredByOf(replies, { attestedGen = false, reportedAll = new Set() } = {}) {
98
+ const kinds = new Set(replies.map((r) => r && r.answeredBy).filter(Boolean));
99
+ const kind = replies.length ? (kinds.size === 1 && kinds.has('stub') ? 'stub' : 'model') : (stubEnabled() ? 'stub' : 'model');
100
+ const iso = replies.map((r) => r && r.isolation).find(Boolean) || 'none';
101
+ const reportedModels = reportedAll.size ? [...reportedAll].sort() : null;
102
+ return { kind, attested: kind === 'model' && attestedGen, reported_models: reportedModels, isolation: iso };
65
103
  }
66
104
 
67
105
  // Determine a case outcome from its sampled band and threshold.
@@ -80,6 +118,12 @@ function outcomeFor(mean, stddev, threshold) {
80
118
  // under --keep-transcripts; never part of the receipt).
81
119
  async function judgeCase({ caseObj, response, generationHash, judgeModel, mode, timeoutMs, samples, trusted = false }) {
82
120
  const g = await gradeSamples({ task: caseObj.prompt, response, rubric: caseObj.rubric, model: judgeModel, samples, timeoutMs, trusted });
121
+ // A judge that produced no score produced no measurement (spec 026 AC-3):
122
+ // the draw is unmeasured with the judge's reason, and no sample is kept.
123
+ // The judge calls that WERE made travel with the unmeasured result: their
124
+ // output hashes (transcript auditability: the judge said something, and a
125
+ // reader can check what) and their usage, so the receipt records every call.
126
+ if (g.unmeasured) return { unmeasured: true, reason: g.reason, sampleTexts: g.sample_texts, sampleHashes: g.sample_hashes || [], attempts: g.attempts, replies: g.replies || [], judge_usage: hasUsage(g.usage) ? g.usage : null };
83
127
  const outcome = outcomeFor(g.mean, g.stddev, caseObj.pass_threshold);
84
128
  const caseResult = {
85
129
  id: caseObj.id,
@@ -102,7 +146,7 @@ async function judgeCase({ caseObj, response, generationHash, judgeModel, mode,
102
146
  if (checks.length) caseResult.checks = checks;
103
147
  // v0.4: grading overhead for this case row, kept OUT of the skill-value math.
104
148
  if (hasUsage(g.usage)) caseResult.judge_usage = g.usage;
105
- return { caseResult, sampleTexts: g.sample_texts, attempts: g.attempts };
149
+ return { caseResult, sampleTexts: g.sample_texts, attempts: g.attempts, replies: g.replies || [] };
106
150
  }
107
151
 
108
152
  // Run up to `concurrency` async tasks at a time, preserving input order in the
@@ -155,6 +199,13 @@ const resolveCallTimeoutMs = function resolveCallTimeoutMs(surface, opts = {}) {
155
199
  async function runSkillOnModel({ skill, model, opts = {} }) {
156
200
  const modelId = resolveModel(model);
157
201
  const judgeModel = opts.judgeModel ? resolveModel(opts.judgeModel) : modelId;
202
+ // Spec 026 AC-10 (F6): the target and the judge must be models the registry
203
+ // knows, checked HERE, on the path the CLI, the report scripts and the
204
+ // trigger all share, before the cost guard and before any call. bin/driftproof
205
+ // makes the same check at its door so the refusal names the registry path in
206
+ // its own message; this one is the door every caller passes.
207
+ assertRegistered(modelId, 'model');
208
+ assertRegistered(judgeModel, 'judge model');
158
209
  // THE SURFACE'S OWN DECLARED TIMEOUT, resolved per arm (spec 017 AC-1).
159
210
  //
160
211
  // This read `opts.timeoutMs || 120000`, and `lib/provider.js` documents that an
@@ -204,6 +255,12 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
204
255
 
205
256
  let calls = 0;
206
257
  let failedCases = 0;
258
+ // Spec 026 AC-1, AC-2: every reply the run received, for the answered_by
259
+ // block; whether every generation reply attested the requested model (read
260
+ // from the surface's echo, AC-2); every canonical id any reply named.
261
+ const replies = [];
262
+ let attestedGen = true;
263
+ const reportedAll = new Set();
207
264
  const isTimeout = (e) => !!(e && (e.code === 'TIMEOUT' || /tim(e|ed)\s*out/i.test(String((e && e.message) || ''))));
208
265
  const genKind = (ws) => (ws ? 'gen_with_skill' : 'gen_baseline');
209
266
  const pairs = await mapPool(tasks, concurrency, async ({ c, withSkill }) => {
@@ -221,6 +278,27 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
221
278
  let lastTranscript = null;
222
279
  let action = { stop: false, reason: 'below_min' };
223
280
  let fatal = null;
281
+ // Spec 026 AC-3: an unmeasured draw that was NOT a timeout (an empty
282
+ // generation, a judge with no score) makes the case failed_unmeasured
283
+ // rather than failed_timeout when no draw measured.
284
+ let nonTimeout = false;
285
+
286
+ // An UNMEASURED draw carries no score, no fabricated samples, and the
287
+ // reason it carries none; it is excluded from every statistic rather than
288
+ // counted as a zero. v0.6 adds what the surface said about the reply. The
289
+ // generation's own hash is kept when there was text.
290
+ const unmeasuredDraw = (drawIndex, reason, gen) => ({
291
+ draw_index: drawIndex,
292
+ generation_hash: gen && String(gen.text || '') ? sha256(String(gen.text || '')) : null,
293
+ status: 'unmeasured',
294
+ reason: String(reason || '').slice(0, 200),
295
+ samples: [],
296
+ mean: null,
297
+ stddev: null,
298
+ stop_reason: gen ? (gen.stopReason || null) : null,
299
+ truncated: !!(gen && gen.truncated === true),
300
+ reported_model: gen && gen.attested ? modelId : null,
301
+ });
224
302
 
225
303
  while (!action.stop && draws.length < SAMPLING.max) {
226
304
  const drawIndex = draws.length;
@@ -229,11 +307,64 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
229
307
  const gen = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs: ctGen, trusted });
230
308
  calls += 1;
231
309
  if (budget) budget.add((gen.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
310
+ replies.push(gen);
311
+ // What answered, on canonical ids; a different model stops the run.
312
+ const a = attest(gen, modelId, 'generation');
313
+ gen.attested = a.attested;
314
+ if (!a.attested) attestedGen = false;
315
+ for (const id of (a.reported || [])) reportedAll.add(canonicalModelId(id));
316
+ gen.truncated = isTruncated(gen.stopReason);
317
+ // Spec 026 AC-8 (F4): a generation cut at the output cap is a partial
318
+ // answer, and a partial answer graded as a whole one is a score about
319
+ // something the model did not write. Recorded truncated, unmeasured,
320
+ // never judged.
321
+ if (gen.truncated === true) {
322
+ nonTimeout = true;
323
+ const d = unmeasuredDraw(drawIndex, `generation truncated at the output cap (stop_reason ${gen.stopReason})`, gen);
324
+ if (hasUsage(gen.usage)) d.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
325
+ draws.push(d);
326
+ onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: d.reason });
327
+ action = nextAction(draws);
328
+ continue;
329
+ }
330
+ // Spec 026 AC-3 (F2): an empty generation is the absence of an output,
331
+ // not an output that scored zero. It is not sent to the judge.
332
+ if (!String(gen.text || '').trim()) {
333
+ nonTimeout = true;
334
+ const d = unmeasuredDraw(drawIndex, 'empty generation (the surface returned no text)', gen);
335
+ if (hasUsage(gen.usage)) d.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
336
+ draws.push(d);
337
+ onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: d.reason });
338
+ action = nextAction(draws);
339
+ continue;
340
+ }
232
341
  const generationHash = sha256(String(gen.text || ''));
233
342
  onProgress({ case: c.id, mode, phase: 'judge', samples, draw: drawIndex });
234
343
  const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ctJudge, samples, trusted });
235
344
  calls += samples;
236
345
  if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judgeModel, 'judge'));
346
+ for (const r of (jr.replies || [])) {
347
+ if (!r) continue;
348
+ replies.push(r);
349
+ const ja = attest(r, judgeModel, 'judge');
350
+ for (const id of (ja.reported || [])) reportedAll.add(canonicalModelId(id));
351
+ }
352
+ if (jr.unmeasured) {
353
+ // The judge returned no score for this draw (empty, unparseable,
354
+ // non-numeric, out of range): unmeasured, naming the judge's reason;
355
+ // the generation's own hash is kept.
356
+ nonTimeout = true;
357
+ const d = unmeasuredDraw(drawIndex, jr.reason, gen);
358
+ // The judge samples taken before the draw was called unmeasured are
359
+ // recorded by hash (no score entered any statistic; the calls happened).
360
+ if ((jr.sampleHashes || []).length) d.judge_sample_hashes = jr.sampleHashes;
361
+ if (hasUsage(gen.usage)) d.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
362
+ if (jr.judge_usage) d.judge_usage = jr.judge_usage;
363
+ draws.push(d);
364
+ onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: d.reason });
365
+ action = nextAction(draws);
366
+ continue;
367
+ }
237
368
  const draw = {
238
369
  draw_index: drawIndex,
239
370
  generation_hash: generationHash,
@@ -242,6 +373,12 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
242
373
  judge_sample_hashes: jr.caseResult.judge_sample_hashes,
243
374
  mean: jr.caseResult.mean,
244
375
  stddev: jr.caseResult.stddev,
376
+ // v0.6 (spec 026 AC-2, AC-8): why the generation stopped, that it
377
+ // was not cut (a cut draw never reaches here), and what the surface
378
+ // said served it (null when it said nothing).
379
+ stop_reason: gen.stopReason || null,
380
+ truncated: false,
381
+ reported_model: gen.attested ? modelId : null,
245
382
  };
246
383
  if (hasUsage(gen.usage)) draw.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
247
384
  if (jr.caseResult.judge_usage) draw.judge_usage = jr.caseResult.judge_usage;
@@ -250,7 +387,7 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
250
387
  if (keepTranscripts) lastTranscript = { id: c.id, mode, generation: String(gen.text || ''), judge_outputs: jr.sampleTexts };
251
388
  } catch (e) {
252
389
  if (e && e.code === 'BUDGET_HARDSTOP') throw e; // budget hard-stop stays fatal
253
- if (!isTimeout(e)) { fatal = e; break; } // non-timeout errors stay fatal
390
+ if (!isTimeout(e)) { fatal = e; break; } // non-timeout errors stay fatal (a SUBSTRATE_MISMATCH among them)
254
391
  if (budget) {
255
392
  try {
256
393
  if (e.phase === 'judge') budget.add((e.judgeAttempts || 1) * perCallCostUSD(judgeModel, 'judge'));
@@ -268,6 +405,9 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
268
405
  samples: [],
269
406
  mean: null,
270
407
  stddev: null,
408
+ stop_reason: null,
409
+ truncated: false,
410
+ reported_model: null,
271
411
  });
272
412
  onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: String((e && e.message) || 'timeout') });
273
413
  }
@@ -291,14 +431,22 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
291
431
  // canary was dropped by exactly such an assembly silently gaining a field
292
432
  // upstream that nothing here carried down (F-014-D).
293
433
  variance_ratio_unavailable: agg.variance_ratio_unavailable,
434
+ // v0.6 (spec 026 AC-8): draws cut at the output cap, counted here so a
435
+ // reader sees it without walking the draw list.
436
+ n_truncated: draws.filter((d) => d.truncated === true).length,
294
437
  draws,
295
438
  };
296
439
 
297
- // Every draw failed: the case is recorded failed_timeout as before, but it
298
- // now carries the draw list showing WHAT failed and how often.
440
+ // Every draw failed: the case is recorded failed, carrying the draw list
441
+ // showing WHAT failed and how often. failed_timeout when every failure was
442
+ // a timeout; failed_unmeasured when any draw was unmeasured for another
443
+ // reason (spec 026 AC-3). The two share one predicate, caseFailed, in
444
+ // lib/receipt.js, and no reader excludes by either literal.
299
445
  if (!last) {
300
446
  failedCases += 1;
301
- return { caseResult: { id: c.id, mode, case_status: 'failed_timeout', reason: (draws[draws.length - 1] || {}).reason || 'timeout', generation }, transcript: null };
447
+ const status = nonTimeout ? FAILED_STATUSES[1] : FAILED_STATUSES[0];
448
+ const lastReason = [...draws].reverse().map((d) => d.reason).find(Boolean) || 'timeout';
449
+ return { caseResult: { id: c.id, mode, case_status: status, reason: lastReason, generation }, transcript: null };
302
450
  }
303
451
 
304
452
  // The v0.4-shaped fields now describe the DRAW SET, not one arbitrary draw,
@@ -317,7 +465,14 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
317
465
  const caseResults = pairs.map((p) => p.caseResult);
318
466
  const transcripts = keepTranscripts ? pairs.map((p) => p.transcript) : null;
319
467
 
320
- const surface = surfaceForModel(modelId);
468
+ // THE SURFACE IS WHAT ANSWERED, not what the runner would have chosen for
469
+ // the model id (spec 026 AC-1, F1). A stub run records surface stub, judge
470
+ // surface stub, answered_by.kind stub, and is UNVERIFIED: it measured nothing.
471
+ const answered = answeredByOf(replies, { attestedGen: attestedGen && replies.some((r) => r && r.answeredBy === 'model'), reportedAll });
472
+ const answeredBy = { kind: answered.kind, attested: answered.attested, reported_model: answered.attested ? modelId : null, reported_models: answered.reported_models, isolation: answered.isolation };
473
+ const surface = answered.kind === 'stub' ? 'stub' : surfaceForModel(modelId);
474
+ // v0.6 (spec 026 AC-11): which judge ran, and the template it graded with.
475
+ const judgeBlock = { ...judgeSettings(samples, judgeModel), ...(answered.kind === 'stub' ? { surface: 'stub' } : {}), model_id: judgeModel, prompt_template_hash: promptTemplateHash() };
321
476
  const nowIso = opts.nowIso || new Date().toISOString();
322
477
  // v0.4 economics. The pricing snapshot is frozen HERE, at run time, from the
323
478
  // registry; every derived dollar figure below is computed from the snapshot and
@@ -359,24 +514,52 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
359
514
  model_release_date: releaseDateFor(modelId),
360
515
  provider: providerForModel(modelId),
361
516
  surface,
362
- // v0.3.1: on the openai/cli (codex) surface, record the fixed harness preamble.
517
+ // v0.3.1: on the openai/cli (codex) surface, record the fixed harness
518
+ // preamble. Absent on a stub run: it describes a harness that did not run.
363
519
  surface_overhead_note: surface === 'openai-cli' ? CODEX_OVERHEAD_NOTE : undefined,
364
520
  runner_version: RUNNER_VERSION,
365
521
  date_utc: nowIso,
366
522
  registry: registryStatus(modelId),
367
523
  transcripts: keepTranscripts ? 'retained-local' : 'hashes-only',
368
- judge: judgeSettings(samples, judgeModel),
524
+ judge: judgeBlock,
369
525
  pricing_snapshot: pricingSnapshot,
526
+ answered_by: answeredBy,
370
527
  },
371
528
  cases: caseResults,
372
529
  economics,
373
- verificationLevel: 'TESTED',
530
+ // The level is DERIVED from what answered: only a model-answered run may
531
+ // read TESTED. The schema refuses TESTED on a stub receipt as the second,
532
+ // independent control (spec 026 AC-1).
533
+ verificationLevel: answered.kind === 'model' ? 'TESTED' : 'UNVERIFIED',
374
534
  });
375
535
 
376
536
  return { receipt, calls, transcripts, failedCases };
377
537
  }
378
538
 
379
- function band(mean, sd) { return `${mean.toFixed(3)} ± ${sd.toFixed(3)}`; }
539
+ // A band the formula could not form is printed as what it is, never as 0.000
540
+ // (spec 026 AC-7): one included case has a mean and no dispersion; none has
541
+ // neither.
542
+ function band(mean, sd) {
543
+ if (mean == null) return 'n/a (0 cases)';
544
+ if (sd == null) return `${mean.toFixed(3)} ± n/a (1 case)`;
545
+ return `${mean.toFixed(3)} ± ${sd.toFixed(3)}`;
546
+ }
547
+ // The comparison band, or the reason there is none.
548
+ function uncertaintyStr(cmp) {
549
+ if (cmp.delta_uncertainty != null) return cmp.delta_uncertainty.toFixed(3);
550
+ return `n/a (${cmp.delta_uncertainty_unavailable === 'single_case' ? '1 case' : cmp.delta_uncertainty_unavailable === 'no_cases' ? '0 cases' : 'no band'})`;
551
+ }
552
+ // The answered_by block, said in one line for the summary and the CLI.
553
+ function answeredLine(receipt) {
554
+ const ab = (receipt.run && receipt.run.answered_by) || null;
555
+ if (!ab) return 'answered by: unrecorded (pre-v0.6 receipt)';
556
+ if (ab.kind === 'stub') return 'answered by: stub (DRIFTPROOF_STUB) — nothing answered; this run measured nothing (UNVERIFIED)';
557
+ if (ab.kind === 'external') return 'answered by: an external tool (imported)';
558
+ const iso = ab.isolation === 'eval-user' ? 'the isolated eval-user hop' : ab.isolation === 'same-user' ? 'the same-user spawn (--trusted-skill)' : 'no spawn (api surface)';
559
+ return ab.attested
560
+ ? `answered by: model ${ab.reported_model} (attested by the surface; ${iso})`
561
+ : `answered by: model, but the surface did not report which model answered (attested: false; ${iso})`;
562
+ }
380
563
 
381
564
  // Render a short human-readable markdown summary of a receipt.
382
565
  function summarizeReceipt(receipt) {
@@ -385,6 +568,7 @@ function summarizeReceipt(receipt) {
385
568
  L.push('');
386
569
  L.push(`- **model:** \`${receipt.run.model_id}\`${receipt.run.model_release_date ? ` (released ${receipt.run.model_release_date})` : ''}`);
387
570
  L.push(`- **surface:** ${receipt.run.surface}`);
571
+ L.push(`- **${answeredLine(receipt)}**`);
388
572
  L.push(`- **run (UTC):** ${receipt.run.date_utc}`);
389
573
  L.push(`- **runner:** v${receipt.run.runner_version}`);
390
574
  const j = receipt.run.judge || {};
@@ -399,22 +583,39 @@ function summarizeReceipt(receipt) {
399
583
  L.push('');
400
584
  const cmp = receipt.comparison;
401
585
  const aggs = receipt.results.aggregates;
402
- const sign = cmp.delta >= 0 ? '+' : '';
586
+ if ((receipt.run.answered_by || {}).kind === 'stub') {
587
+ L.push('> **STUB RUN** — DRIFTPROOF_STUB=1: nothing answered, the text was canned, and this run measured nothing. The receipt is UNVERIFIED and verdicts nothing.');
588
+ L.push('');
589
+ }
403
590
  if (receipt.run.status === 'incomplete') {
404
591
  L.push(`> ⏱ **INCOMPLETE** — ${receipt.run.failed_case_count} case(s) had an arm that could not be measured and are EXCLUDED from the aggregates below, BOTH arms together; this receipt must not be used to compute a drift/durability verdict.`);
405
592
  L.push('');
406
593
  }
594
+ // Spec 026 AC-8: draws cut at the output cap, said once for the run.
595
+ const nTruncated = receipt.results.cases.reduce((a, c) => a + (((c.generation || {}).n_truncated) || 0), 0);
596
+ if (nTruncated) {
597
+ L.push(`> ✂ **${nTruncated} draw(s) truncated** at the output cap: each is unmeasured and excluded from every band below.`);
598
+ L.push('');
599
+ }
407
600
  L.push(`with_skill **${band(aggs.with_skill.mean_score, aggs.with_skill.stddev)}** vs baseline **${band(aggs.baseline.mean_score, aggs.baseline.stddev)}**`);
408
601
  L.push('');
409
- L.push(`skill lift **${sign}${cmp.delta.toFixed(3)}** (combined uncertainty ± ${cmp.delta_uncertainty.toFixed(3)})`);
602
+ if (cmp.delta == null) {
603
+ L.push(`skill lift **n/a** (${cmp.delta_uncertainty_unavailable === 'no_cases' ? 'no case was included on an arm' : 'no comparison'})`);
604
+ } else {
605
+ const sign = cmp.delta >= 0 ? '+' : '';
606
+ L.push(`skill lift **${sign}${cmp.delta.toFixed(3)}** (combined uncertainty ± ${uncertaintyStr(cmp)})`);
607
+ }
608
+ L.push('');
609
+ // The rule the bands above are derived by, said beside them (spec 026 AC-6).
610
+ L.push(`band rule: each arm's band is the sample stddev of its per-case means${aggs.band_rule ? ` — ${aggs.band_rule}` : ' (unstated on this pre-v0.6 receipt)'}`);
410
611
  L.push('');
411
612
  L.push(`## Per-case (mean ± stddev over ${(receipt.run.judge || {}).samples || 1} judge samples)`);
412
613
  L.push('');
413
614
  L.push(`| case | mode | outcome | mean ± stddev | judge reason |`);
414
615
  L.push(`|---|---|---|---|---|`);
415
616
  for (const c of receipt.results.cases) {
416
- if (c.case_status === 'failed_timeout') {
417
- L.push(`| \`${c.id}\` | ${c.mode} | ⏱ failed_timeout | — (not measured) | ${c.reason || 'timed out'} |`);
617
+ if (caseFailed(c)) {
618
+ L.push(`| \`${c.id}\` | ${c.mode} | ⏱ ${c.case_status} | — (not measured) | ${c.reason || 'not measured'} |`);
418
619
  continue;
419
620
  }
420
621
  const flag = c.outcome === 'borderline' ? ' ⚠' : '';
@@ -424,4 +625,4 @@ function summarizeReceipt(receipt) {
424
625
  return L.join('\n');
425
626
  }
426
627
 
427
- module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor, resolveCallTimeoutMs };
628
+ module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor, resolveCallTimeoutMs, answeredLine, answeredByOf, attest, band, uncertaintyStr };
package/lib/skill.js CHANGED
@@ -4,7 +4,16 @@
4
4
  const fs = require('fs');
5
5
  const path = require('path');
6
6
  const { sha256Files, sha256Canonical } = require('./canonical');
7
- const { SUITE_FORMAT } = require('../config');
7
+ const { SUITE_FORMAT, SKILL_MAX_FILES, SKILL_MAX_BYTES, SKILL_MAX_DEPTH, SUITE_MAX_CASES, CASE_MAX_CHARS } = require('../config');
8
+
9
+ // A bound was exceeded: the error names the bound and the value (spec 026
10
+ // AC-13), so the refusal says what to change rather than that something is
11
+ // too big.
12
+ function pastBound(bound, limit, value, what) {
13
+ const e = new Error(`${what}: ${value} exceeds ${bound} (${limit}); refusing to load`);
14
+ e.code = 'INPUT_BOUND'; e.bound = bound; e.limit = limit; e.value = value;
15
+ return e;
16
+ }
8
17
 
9
18
  // Load a skill directory and its eval suite.
10
19
  //
@@ -21,13 +30,22 @@ const { SUITE_FORMAT } = require('../config');
21
30
  const IGNORE_DIRS = new Set(['.git', 'node_modules', 'evals']);
22
31
  const IGNORE_FILES = new Set(['.DS_Store']);
23
32
 
24
- function walkFiles(dir, base = dir, acc = []) {
33
+ // Bounded (spec 026 AC-13): the walk stops at SKILL_MAX_DEPTH directories
34
+ // below the skill dir, SKILL_MAX_FILES bundled files, and SKILL_MAX_BYTES of
35
+ // them together, and throws naming the bound the moment one is passed, so a
36
+ // pathological tree is refused before its bytes are read into memory.
37
+ function walkFiles(dir, base = dir, acc = [], state = { bytes: 0 }, depth = 0) {
38
+ if (depth > SKILL_MAX_DEPTH) throw pastBound('SKILL_MAX_DEPTH', SKILL_MAX_DEPTH, depth, `directory depth under ${base}`);
25
39
  for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
26
40
  if (entry.isDirectory()) {
27
41
  if (IGNORE_DIRS.has(entry.name)) continue;
28
- walkFiles(path.join(dir, entry.name), base, acc);
42
+ walkFiles(path.join(dir, entry.name), base, acc, state, depth + 1);
29
43
  } else if (entry.isFile() && !IGNORE_FILES.has(entry.name)) {
30
44
  const abs = path.join(dir, entry.name);
45
+ if (acc.length + 1 > SKILL_MAX_FILES) throw pastBound('SKILL_MAX_FILES', SKILL_MAX_FILES, acc.length + 1, `bundled files under ${base}`);
46
+ const size = fs.statSync(abs).size;
47
+ state.bytes += size;
48
+ if (state.bytes > SKILL_MAX_BYTES) throw pastBound('SKILL_MAX_BYTES', SKILL_MAX_BYTES, state.bytes, `bundled bytes under ${base}`);
31
49
  acc.push({ path: path.relative(base, abs), bytes: fs.readFileSync(abs) });
32
50
  }
33
51
  }
@@ -108,12 +126,16 @@ function normalizeCases(raw) {
108
126
  : Array.isArray(raw.evals) ? raw.evals
109
127
  : null;
110
128
  if (!list) throw new Error('evals.json must be an array or have a `cases`/`evals` array');
129
+ // Bounded (spec 026 AC-13): the case count, and each prompt and rubric.
130
+ if (list.length > SUITE_MAX_CASES) throw pastBound('SUITE_MAX_CASES', SUITE_MAX_CASES, list.length, 'cases in the suite');
111
131
  return list.map((c, i) => {
112
132
  const id = String(c.id || c.name || `case-${i + 1}`);
113
133
  const prompt = c.prompt || c.input || c.task;
114
134
  const rubric = c.rubric || c.criteria || c.expected;
115
135
  if (!prompt) throw new Error(`case "${id}" is missing a prompt/input/task`);
116
136
  if (!rubric) throw new Error(`case "${id}" is missing a rubric/criteria/expected`);
137
+ if (String(prompt).length > CASE_MAX_CHARS) throw pastBound('CASE_MAX_CHARS', CASE_MAX_CHARS, String(prompt).length, `case "${id}" prompt characters`);
138
+ if (String(rubric).length > CASE_MAX_CHARS) throw pastBound('CASE_MAX_CHARS', CASE_MAX_CHARS, String(rubric).length, `case "${id}" rubric characters`);
117
139
  const threshold = typeof c.pass_threshold === 'number' ? c.pass_threshold
118
140
  : typeof c.threshold === 'number' ? c.threshold : 0.7;
119
141
  const norm = { id, prompt: String(prompt), rubric: String(rubric), pass_threshold: threshold };
@@ -124,4 +146,4 @@ function normalizeCases(raw) {
124
146
  });
125
147
  }
126
148
 
127
- module.exports = { loadSkill, normalizeCases, parseSkillMeta };
149
+ module.exports = { loadSkill, normalizeCases, parseSkillMeta, pastBound };
package/lib/stats.js CHANGED
@@ -31,8 +31,11 @@ function stderr(xs) {
31
31
  return round(stddev(xs) / Math.sqrt(n));
32
32
  }
33
33
 
34
- // Combine independent uncertainties in quadrature: sqrt(a^2 + b^2).
34
+ // Combine independent uncertainties in quadrature: sqrt(a^2 + b^2). Null when
35
+ // either band is null: a combination of a band that could not form cannot form
36
+ // either (spec 026 AC-7), and the receipt says why beside it.
35
37
  function combineUncertainty(a, b) {
38
+ if (a == null || b == null) return null;
36
39
  return round(Math.sqrt(a * a + b * b));
37
40
  }
38
41
 
@@ -45,10 +48,16 @@ function combineUncertainty(a, b) {
45
48
  // is a conventional, honest "mean ± stddev across the suite". The drift HEADLINE
46
49
  // verdict is driven by the per-case band-overlap verdicts (see lib/diff.js), not
47
50
  // by this aggregate band; this value is a reported summary statistic.
51
+ //
52
+ // A BAND THE FORMULA CANNOT FORM IS NULL, NEVER 0 (spec 026 AC-7, F3). The
53
+ // sample standard deviation of one value is undefined, and the mean of no
54
+ // values is not a number; printing 0.000 for either asserted a precision that
55
+ // was never measured (the one-case run's `± 0.000`). The rule the receipt
56
+ // states (results.aggregates.band_rule) is this function.
48
57
  function aggregateBands(cases) {
49
- if (!cases.length) return { mean: 0, stddev: 0 };
58
+ if (!cases.length) return { mean: null, stddev: null };
50
59
  const means = cases.map((c) => c.mean);
51
- return { mean: mean(means), stddev: stddev(means) };
60
+ return { mean: mean(means), stddev: means.length < 2 ? null : stddev(means) };
52
61
  }
53
62
 
54
63
  // Do two confidence bands (mean ± half-width) fail to overlap, and in which
package/lib/usage.js CHANGED
@@ -42,6 +42,27 @@
42
42
 
43
43
  function n(x) { return Number.isFinite(Number(x)) ? Number(x) : 0; }
44
44
 
45
+ // ── what answered (spec 026, AC-2 and AC-8) ──────────────────────────────────
46
+ // Every lane also reports, when it can, WHICH model served the call and WHY the
47
+ // reply stopped. Both are read here, never guessed: a surface that says nothing
48
+ // yields null for both, and the receipt records that it said nothing.
49
+ //
50
+ // reportedModels the ids the surface named, VERBATIM (the real claude CLI
51
+ // keys the call's usage under the undated form and puts a
52
+ // dated side entry beside it; an api response names one
53
+ // model). Null when the surface names no model. The runner
54
+ // compares on canonical ids (canonicalModelId: a trailing
55
+ // -YYYYMMDD is not part of the identity), never the reader.
56
+ // stopReason the surface's own word for why the reply ended (end_turn,
57
+ // max_tokens, stop, length ...). Null when it gives none.
58
+ function canonicalModelId(id) { return String(id || '').replace(/-\d{8}$/, ''); }
59
+ function reportedModelsOf(map) {
60
+ if (!map || typeof map !== 'object' || Array.isArray(map)) return null;
61
+ const ids = [...new Set(Object.keys(map).filter((k) => k))].sort();
62
+ return ids.length ? ids : null;
63
+ }
64
+ function stopReasonOf(v) { return typeof v === 'string' && v ? v : null; }
65
+
45
66
  // The empty/unknown usage record. Deliberately null (not zeros) so "the surface
46
67
  // did not tell us" never reads as "the call cost nothing".
47
68
  function emptyUsage() {
@@ -75,6 +96,9 @@ function parseClaudeCliJson(stdout) {
75
96
  text: typeof j.result === 'string' ? j.result : '',
76
97
  usage,
77
98
  isError: j.is_error === true,
99
+ // v0.6: the CLI's own stop reason and the models it says served the call.
100
+ stopReason: stopReasonOf(j.stop_reason),
101
+ reportedModels: reportedModelsOf(j.modelUsage),
78
102
  // The CLI reports its own dollar figure. Recorded here for completeness but
79
103
  // NOT used: costs are computed uniformly from the frozen pricing snapshot so
80
104
  // three substrates are on one basis (see lib/value.js).
@@ -85,7 +109,9 @@ function parseClaudeCliJson(stdout) {
85
109
  // ── openai/cli — `codex exec --json` JSONL event stream ───────────────────────
86
110
  // One JSON object per line. Usage rides the terminal `turn.completed` event;
87
111
  // input_tokens already INCLUDES cached_input_tokens. Unparseable lines are
88
- // skipped (the stream also carries progress events we do not model).
112
+ // skipped (the stream also carries progress events we do not model). The
113
+ // stream names neither the model that served the turn nor a stop reason, so
114
+ // both read null here and the receipt says the surface did not report them.
89
115
  function parseCodexJsonl(stdout) {
90
116
  const lines = String(stdout || '').split('\n').map((l) => l.trim()).filter(Boolean);
91
117
  let usage = null;
@@ -103,16 +129,30 @@ function parseCodexJsonl(stdout) {
103
129
  wall_ms: null,
104
130
  };
105
131
  }
106
- // The final message is normally read from the -o file; this is a fallback.
132
+ // The final message IS the stream: the last `item.completed` /
133
+ // `agent_message` event carries it. Nothing is read from a file after the
134
+ // call, on either spawn mode (spec 022 AC-9; the earlier comment here
135
+ // described an output file the lane no longer reads, F-022-3).
107
136
  if (ev.type === 'item.completed' && ev.item && ev.item.type === 'agent_message' && typeof ev.item.text === 'string') {
108
137
  text = ev.item.text;
109
138
  }
110
139
  }
111
- return { usage, text };
140
+ return { usage, text, stopReason: null, reportedModels: null };
112
141
  }
113
142
 
114
143
  // ── anthropic/api ─────────────────────────────────────────────────────────────
115
144
  // The Messages API reports input_tokens EXCLUDING cache, same as the CLI.
145
+ // The response object also names the model that served it (`model`) and why
146
+ // it stopped (`stop_reason`); read here from the object, null when absent.
147
+ // Exercised against tests/fixtures/response-anthropic-api.json, which is
148
+ // SYNTHESISED from the API reference, not captured from a call.
149
+ function readAnthropicApiResponse(resp) {
150
+ const r = resp && typeof resp === 'object' ? resp : {};
151
+ return {
152
+ stopReason: stopReasonOf(r.stop_reason),
153
+ reportedModels: typeof r.model === 'string' && r.model ? [r.model] : null,
154
+ };
155
+ }
116
156
  function parseAnthropicApiUsage(u) {
117
157
  if (!u) return emptyUsage();
118
158
  const cacheRead = n(u.cache_read_input_tokens);
@@ -127,7 +167,18 @@ function parseAnthropicApiUsage(u) {
127
167
 
128
168
  // ── openai/api (Chat Completions-compatible) ──────────────────────────────────
129
169
  // prompt_tokens is the total; the cached portion, when present, is nested under
130
- // prompt_tokens_details.cached_tokens.
170
+ // prompt_tokens_details.cached_tokens. The response also names the serving
171
+ // model (`model`) and the first choice's `finish_reason`; read here, null
172
+ // when absent. Exercised against tests/fixtures/response-openai-api.json,
173
+ // SYNTHESISED from the API reference, not captured from a call.
174
+ function readOpenaiApiResponse(resp) {
175
+ const r = resp && typeof resp === 'object' ? resp : {};
176
+ const choice = Array.isArray(r.choices) && r.choices[0] && typeof r.choices[0] === 'object' ? r.choices[0] : {};
177
+ return {
178
+ stopReason: stopReasonOf(choice.finish_reason),
179
+ reportedModels: typeof r.model === 'string' && r.model ? [r.model] : null,
180
+ };
181
+ }
131
182
  function parseOpenaiApiUsage(u) {
132
183
  if (!u) return emptyUsage();
133
184
  const details = u.prompt_tokens_details || {};
@@ -165,4 +216,5 @@ function normalizeUsage(u) {
165
216
  module.exports = {
166
217
  emptyUsage, hasUsage, sumUsage, normalizeUsage,
167
218
  parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage,
219
+ readAnthropicApiResponse, readOpenaiApiResponse, canonicalModelId, reportedModelsOf,
168
220
  };
package/lib/value.js CHANGED
@@ -1,6 +1,8 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
+ const { caseFailed } = require('./receipt');
5
+
4
6
  const { EFFECT_FLOOR } = require('../config');
5
7
 
6
8
  // Economics of a run — what a skill COSTS to run, alongside whether it helps.
@@ -204,7 +206,7 @@ function armEconomics(rows, price) {
204
206
  // surface the run surface, to label the cost basis honestly
205
207
  function computeEconomics({ cases, modelId, judgeModelId, pricingSnapshot, surface, meteredSurface }) {
206
208
  const price = (pricingSnapshot && pricingSnapshot.models && pricingSnapshot.models[modelId]) || null;
207
- const ok = (cases || []).filter((c) => c.case_status !== 'failed_timeout');
209
+ const ok = (cases || []).filter((c) => !caseFailed(c));
208
210
  const withRows = ok.filter((c) => c.mode === 'with_skill');
209
211
  const baseRows = ok.filter((c) => c.mode === 'baseline');
210
212
 
@@ -322,7 +324,7 @@ function receiptCostBreakdown(receipt) {
322
324
  let judge = 0;
323
325
  let missingJudgeRate = null;
324
326
  for (const c of ((receipt.results || {}).cases || [])) {
325
- if (c.case_status === 'failed_timeout') continue;
327
+ if (caseFailed(c)) continue;
326
328
  const g = costForUsage(c.usage, genRate);
327
329
  if (g != null) generation += g;
328
330
  if (c.judge_usage) {
package/lib/verdict.js CHANGED
@@ -43,7 +43,12 @@ function verdictFromReceipt(receipt) {
43
43
  // Below-TESTED receipts (imported/DECLARED) and null-delta receipts are never
44
44
  // verdicted — we did not run the suite, so we do not certify the outcome.
45
45
  const level = (receipt && receipt.verification_level) || 'TESTED';
46
- const measured = level === 'TESTED' && typeof cmp.delta === 'number';
46
+ // Spec 026 AC-4: an INCOMPLETE receipt (its run block's status field reads
47
+ // incomplete: a case had an arm that could not be measured) is not verdicted
48
+ // either. spec/RECEIPT.md has said since v0.3.1 that a drift report must not
49
+ // compute a verdict from one; the badge is the same reader with a shorter path.
50
+ const incomplete = !!(receipt && receipt.run && receipt.run.status === 'incomplete');
51
+ const measured = level === 'TESTED' && typeof cmp.delta === 'number' && !incomplete;
47
52
  let verdict;
48
53
  if (!measured) verdict = 'NOT_MEASURED';
49
54
  else if (delta >= EFFECT_FLOOR) verdict = 'PASSED';