driftproof 0.8.1 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +87 -16
- package/bin/driftproof +289 -24
- package/config.js +26 -2
- package/lib/checks.js +28 -6
- package/lib/decision.js +425 -0
- package/lib/diff.js +58 -4
- package/lib/importers.js +8 -3
- package/lib/judge.js +69 -12
- package/lib/models.js +52 -11
- package/lib/provider.js +35 -11
- package/lib/receipt.js +51 -11
- package/lib/run.js +221 -20
- package/lib/runner.js +35 -3
- package/lib/skill.js +26 -4
- package/lib/stats.js +12 -3
- package/lib/usage.js +56 -4
- package/lib/value.js +4 -2
- package/lib/verdict.js +6 -1
- package/package.json +3 -2
- package/spec/RECEIPT.md +68 -11
- package/spec/receipt.schema.json +299 -54
- package/spec/receipt.v0.5.schema.json +1318 -0
package/lib/run.js
CHANGED
|
@@ -2,16 +2,18 @@
|
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
4
|
const { complete, resolveModel, surfaceForModel, isMeteredSurface, retryPolicyForSurface, CODEX_OVERHEAD_NOTE } = require('./provider');
|
|
5
|
-
const { gradeSamples, judgeSettings } = require('./judge');
|
|
6
|
-
const { buildReceipt } = require('./receipt');
|
|
5
|
+
const { gradeSamples, judgeSettings, promptTemplateHash, isTruncated } = require('./judge');
|
|
6
|
+
const { buildReceipt, caseFailed, FAILED_STATUSES } = require('./receipt');
|
|
7
7
|
const { sha256 } = require('./canonical');
|
|
8
|
-
const { registryStatus, providerForModel, priceForModel } = require('./models');
|
|
8
|
+
const { registryStatus, providerForModel, priceForModel, assertRegistered } = require('./models');
|
|
9
9
|
const { perCallCostUSD } = require('./cost');
|
|
10
10
|
const { runChecks } = require('./checks');
|
|
11
11
|
const { estimateTokens } = require('./skillCost');
|
|
12
12
|
const { hasUsage, normalizeUsage } = require('./usage');
|
|
13
13
|
const { buildPricingSnapshot, computeEconomics } = require('./value');
|
|
14
14
|
const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES, DEV_MAX_CALLS } = require('../config');
|
|
15
|
+
const { stubEnabled } = require('./stub');
|
|
16
|
+
const { canonicalModelId } = require('./usage');
|
|
15
17
|
const { SAMPLING, acrossDraws, nextAction } = require('./sampling');
|
|
16
18
|
const { suiteCanary } = require('./canary');
|
|
17
19
|
|
|
@@ -60,8 +62,44 @@ async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs, trusted
|
|
|
60
62
|
throw e;
|
|
61
63
|
}
|
|
62
64
|
const system = withSkill ? skillMd : undefined;
|
|
63
|
-
const
|
|
64
|
-
|
|
65
|
+
const out = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs, trusted });
|
|
66
|
+
// The whole reply travels: what answered, what it said served the call, why
|
|
67
|
+
// it stopped, and which spawn path was taken (spec 026 AC-1, AC-2, AC-8).
|
|
68
|
+
return { text: out.text, usage: out.usage, wall_ms: out.wall_ms, attempts: out.attempts, answeredBy: out.answeredBy, surface: out.surface, reportedModels: out.reportedModels, stopReason: out.stopReason, isolation: out.isolation };
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
// ── what answered (spec 026, AC-2) ───────────────────────────────────────────
|
|
72
|
+
// The surface's echo is compared with the requested id on CANONICAL ids (the
|
|
73
|
+
// real claude CLI keys its usage by the undated form). A surface that names a
|
|
74
|
+
// DIFFERENT model stops the run before the next call; no receipt is written,
|
|
75
|
+
// because a receipt naming a model that did not answer is the defect this
|
|
76
|
+
// exists to close. A surface that names nothing is recorded as attested: false.
|
|
77
|
+
function attest(reply, requestedId, phase) {
|
|
78
|
+
const reported = Array.isArray(reply && reply.reportedModels) ? reply.reportedModels : null;
|
|
79
|
+
if (!reported || !reported.length) return { attested: false, reported };
|
|
80
|
+
const want = canonicalModelId(resolveModel(requestedId));
|
|
81
|
+
const other = reported.find((id) => canonicalModelId(id) !== want);
|
|
82
|
+
if (other) {
|
|
83
|
+
const e = new Error(`substrate mismatch: the ${phase} surface answered as "${other}" where "${requestedId}" (canonical "${want}") was requested; the run stops before the next call and no receipt is written`);
|
|
84
|
+
e.code = 'SUBSTRATE_MISMATCH';
|
|
85
|
+
e.requested = requestedId; e.reported = other;
|
|
86
|
+
throw e;
|
|
87
|
+
}
|
|
88
|
+
return { attested: true, reported };
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
// The run's answered_by block, derived from every reply the run received and
|
|
92
|
+
// never from the surface the runner would have chosen (spec 026 AC-1: a stub
|
|
93
|
+
// run used to record the real surface name because the name was computed
|
|
94
|
+
// from the model id, not from what answered). With no reply at all (every
|
|
95
|
+
// call failed before answering) the kind is what the process would have
|
|
96
|
+
// answered with, which is the one thing still known.
|
|
97
|
+
function answeredByOf(replies, { attestedGen = false, reportedAll = new Set() } = {}) {
|
|
98
|
+
const kinds = new Set(replies.map((r) => r && r.answeredBy).filter(Boolean));
|
|
99
|
+
const kind = replies.length ? (kinds.size === 1 && kinds.has('stub') ? 'stub' : 'model') : (stubEnabled() ? 'stub' : 'model');
|
|
100
|
+
const iso = replies.map((r) => r && r.isolation).find(Boolean) || 'none';
|
|
101
|
+
const reportedModels = reportedAll.size ? [...reportedAll].sort() : null;
|
|
102
|
+
return { kind, attested: kind === 'model' && attestedGen, reported_models: reportedModels, isolation: iso };
|
|
65
103
|
}
|
|
66
104
|
|
|
67
105
|
// Determine a case outcome from its sampled band and threshold.
|
|
@@ -80,6 +118,12 @@ function outcomeFor(mean, stddev, threshold) {
|
|
|
80
118
|
// under --keep-transcripts; never part of the receipt).
|
|
81
119
|
async function judgeCase({ caseObj, response, generationHash, judgeModel, mode, timeoutMs, samples, trusted = false }) {
|
|
82
120
|
const g = await gradeSamples({ task: caseObj.prompt, response, rubric: caseObj.rubric, model: judgeModel, samples, timeoutMs, trusted });
|
|
121
|
+
// A judge that produced no score produced no measurement (spec 026 AC-3):
|
|
122
|
+
// the draw is unmeasured with the judge's reason, and no sample is kept.
|
|
123
|
+
// The judge calls that WERE made travel with the unmeasured result: their
|
|
124
|
+
// output hashes (transcript auditability: the judge said something, and a
|
|
125
|
+
// reader can check what) and their usage, so the receipt records every call.
|
|
126
|
+
if (g.unmeasured) return { unmeasured: true, reason: g.reason, sampleTexts: g.sample_texts, sampleHashes: g.sample_hashes || [], attempts: g.attempts, replies: g.replies || [], judge_usage: hasUsage(g.usage) ? g.usage : null };
|
|
83
127
|
const outcome = outcomeFor(g.mean, g.stddev, caseObj.pass_threshold);
|
|
84
128
|
const caseResult = {
|
|
85
129
|
id: caseObj.id,
|
|
@@ -102,7 +146,7 @@ async function judgeCase({ caseObj, response, generationHash, judgeModel, mode,
|
|
|
102
146
|
if (checks.length) caseResult.checks = checks;
|
|
103
147
|
// v0.4: grading overhead for this case row, kept OUT of the skill-value math.
|
|
104
148
|
if (hasUsage(g.usage)) caseResult.judge_usage = g.usage;
|
|
105
|
-
return { caseResult, sampleTexts: g.sample_texts, attempts: g.attempts };
|
|
149
|
+
return { caseResult, sampleTexts: g.sample_texts, attempts: g.attempts, replies: g.replies || [] };
|
|
106
150
|
}
|
|
107
151
|
|
|
108
152
|
// Run up to `concurrency` async tasks at a time, preserving input order in the
|
|
@@ -155,6 +199,13 @@ const resolveCallTimeoutMs = function resolveCallTimeoutMs(surface, opts = {}) {
|
|
|
155
199
|
async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
156
200
|
const modelId = resolveModel(model);
|
|
157
201
|
const judgeModel = opts.judgeModel ? resolveModel(opts.judgeModel) : modelId;
|
|
202
|
+
// Spec 026 AC-10 (F6): the target and the judge must be models the registry
|
|
203
|
+
// knows, checked HERE, on the path the CLI, the report scripts and the
|
|
204
|
+
// trigger all share, before the cost guard and before any call. bin/driftproof
|
|
205
|
+
// makes the same check at its door so the refusal names the registry path in
|
|
206
|
+
// its own message; this one is the door every caller passes.
|
|
207
|
+
assertRegistered(modelId, 'model');
|
|
208
|
+
assertRegistered(judgeModel, 'judge model');
|
|
158
209
|
// THE SURFACE'S OWN DECLARED TIMEOUT, resolved per arm (spec 017 AC-1).
|
|
159
210
|
//
|
|
160
211
|
// This read `opts.timeoutMs || 120000`, and `lib/provider.js` documents that an
|
|
@@ -204,6 +255,12 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
204
255
|
|
|
205
256
|
let calls = 0;
|
|
206
257
|
let failedCases = 0;
|
|
258
|
+
// Spec 026 AC-1, AC-2: every reply the run received, for the answered_by
|
|
259
|
+
// block; whether every generation reply attested the requested model (read
|
|
260
|
+
// from the surface's echo, AC-2); every canonical id any reply named.
|
|
261
|
+
const replies = [];
|
|
262
|
+
let attestedGen = true;
|
|
263
|
+
const reportedAll = new Set();
|
|
207
264
|
const isTimeout = (e) => !!(e && (e.code === 'TIMEOUT' || /tim(e|ed)\s*out/i.test(String((e && e.message) || ''))));
|
|
208
265
|
const genKind = (ws) => (ws ? 'gen_with_skill' : 'gen_baseline');
|
|
209
266
|
const pairs = await mapPool(tasks, concurrency, async ({ c, withSkill }) => {
|
|
@@ -221,6 +278,27 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
221
278
|
let lastTranscript = null;
|
|
222
279
|
let action = { stop: false, reason: 'below_min' };
|
|
223
280
|
let fatal = null;
|
|
281
|
+
// Spec 026 AC-3: an unmeasured draw that was NOT a timeout (an empty
|
|
282
|
+
// generation, a judge with no score) makes the case failed_unmeasured
|
|
283
|
+
// rather than failed_timeout when no draw measured.
|
|
284
|
+
let nonTimeout = false;
|
|
285
|
+
|
|
286
|
+
// An UNMEASURED draw carries no score, no fabricated samples, and the
|
|
287
|
+
// reason it carries none; it is excluded from every statistic rather than
|
|
288
|
+
// counted as a zero. v0.6 adds what the surface said about the reply. The
|
|
289
|
+
// generation's own hash is kept when there was text.
|
|
290
|
+
const unmeasuredDraw = (drawIndex, reason, gen) => ({
|
|
291
|
+
draw_index: drawIndex,
|
|
292
|
+
generation_hash: gen && String(gen.text || '') ? sha256(String(gen.text || '')) : null,
|
|
293
|
+
status: 'unmeasured',
|
|
294
|
+
reason: String(reason || '').slice(0, 200),
|
|
295
|
+
samples: [],
|
|
296
|
+
mean: null,
|
|
297
|
+
stddev: null,
|
|
298
|
+
stop_reason: gen ? (gen.stopReason || null) : null,
|
|
299
|
+
truncated: !!(gen && gen.truncated === true),
|
|
300
|
+
reported_model: gen && gen.attested ? modelId : null,
|
|
301
|
+
});
|
|
224
302
|
|
|
225
303
|
while (!action.stop && draws.length < SAMPLING.max) {
|
|
226
304
|
const drawIndex = draws.length;
|
|
@@ -229,11 +307,64 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
229
307
|
const gen = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs: ctGen, trusted });
|
|
230
308
|
calls += 1;
|
|
231
309
|
if (budget) budget.add((gen.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
|
|
310
|
+
replies.push(gen);
|
|
311
|
+
// What answered, on canonical ids; a different model stops the run.
|
|
312
|
+
const a = attest(gen, modelId, 'generation');
|
|
313
|
+
gen.attested = a.attested;
|
|
314
|
+
if (!a.attested) attestedGen = false;
|
|
315
|
+
for (const id of (a.reported || [])) reportedAll.add(canonicalModelId(id));
|
|
316
|
+
gen.truncated = isTruncated(gen.stopReason);
|
|
317
|
+
// Spec 026 AC-8 (F4): a generation cut at the output cap is a partial
|
|
318
|
+
// answer, and a partial answer graded as a whole one is a score about
|
|
319
|
+
// something the model did not write. Recorded truncated, unmeasured,
|
|
320
|
+
// never judged.
|
|
321
|
+
if (gen.truncated === true) {
|
|
322
|
+
nonTimeout = true;
|
|
323
|
+
const d = unmeasuredDraw(drawIndex, `generation truncated at the output cap (stop_reason ${gen.stopReason})`, gen);
|
|
324
|
+
if (hasUsage(gen.usage)) d.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
|
|
325
|
+
draws.push(d);
|
|
326
|
+
onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: d.reason });
|
|
327
|
+
action = nextAction(draws);
|
|
328
|
+
continue;
|
|
329
|
+
}
|
|
330
|
+
// Spec 026 AC-3 (F2): an empty generation is the absence of an output,
|
|
331
|
+
// not an output that scored zero. It is not sent to the judge.
|
|
332
|
+
if (!String(gen.text || '').trim()) {
|
|
333
|
+
nonTimeout = true;
|
|
334
|
+
const d = unmeasuredDraw(drawIndex, 'empty generation (the surface returned no text)', gen);
|
|
335
|
+
if (hasUsage(gen.usage)) d.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
|
|
336
|
+
draws.push(d);
|
|
337
|
+
onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: d.reason });
|
|
338
|
+
action = nextAction(draws);
|
|
339
|
+
continue;
|
|
340
|
+
}
|
|
232
341
|
const generationHash = sha256(String(gen.text || ''));
|
|
233
342
|
onProgress({ case: c.id, mode, phase: 'judge', samples, draw: drawIndex });
|
|
234
343
|
const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ctJudge, samples, trusted });
|
|
235
344
|
calls += samples;
|
|
236
345
|
if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judgeModel, 'judge'));
|
|
346
|
+
for (const r of (jr.replies || [])) {
|
|
347
|
+
if (!r) continue;
|
|
348
|
+
replies.push(r);
|
|
349
|
+
const ja = attest(r, judgeModel, 'judge');
|
|
350
|
+
for (const id of (ja.reported || [])) reportedAll.add(canonicalModelId(id));
|
|
351
|
+
}
|
|
352
|
+
if (jr.unmeasured) {
|
|
353
|
+
// The judge returned no score for this draw (empty, unparseable,
|
|
354
|
+
// non-numeric, out of range): unmeasured, naming the judge's reason;
|
|
355
|
+
// the generation's own hash is kept.
|
|
356
|
+
nonTimeout = true;
|
|
357
|
+
const d = unmeasuredDraw(drawIndex, jr.reason, gen);
|
|
358
|
+
// The judge samples taken before the draw was called unmeasured are
|
|
359
|
+
// recorded by hash (no score entered any statistic; the calls happened).
|
|
360
|
+
if ((jr.sampleHashes || []).length) d.judge_sample_hashes = jr.sampleHashes;
|
|
361
|
+
if (hasUsage(gen.usage)) d.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
|
|
362
|
+
if (jr.judge_usage) d.judge_usage = jr.judge_usage;
|
|
363
|
+
draws.push(d);
|
|
364
|
+
onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: d.reason });
|
|
365
|
+
action = nextAction(draws);
|
|
366
|
+
continue;
|
|
367
|
+
}
|
|
237
368
|
const draw = {
|
|
238
369
|
draw_index: drawIndex,
|
|
239
370
|
generation_hash: generationHash,
|
|
@@ -242,6 +373,12 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
242
373
|
judge_sample_hashes: jr.caseResult.judge_sample_hashes,
|
|
243
374
|
mean: jr.caseResult.mean,
|
|
244
375
|
stddev: jr.caseResult.stddev,
|
|
376
|
+
// v0.6 (spec 026 AC-2, AC-8): why the generation stopped, that it
|
|
377
|
+
// was not cut (a cut draw never reaches here), and what the surface
|
|
378
|
+
// said served it (null when it said nothing).
|
|
379
|
+
stop_reason: gen.stopReason || null,
|
|
380
|
+
truncated: false,
|
|
381
|
+
reported_model: gen.attested ? modelId : null,
|
|
245
382
|
};
|
|
246
383
|
if (hasUsage(gen.usage)) draw.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
|
|
247
384
|
if (jr.caseResult.judge_usage) draw.judge_usage = jr.caseResult.judge_usage;
|
|
@@ -250,7 +387,7 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
250
387
|
if (keepTranscripts) lastTranscript = { id: c.id, mode, generation: String(gen.text || ''), judge_outputs: jr.sampleTexts };
|
|
251
388
|
} catch (e) {
|
|
252
389
|
if (e && e.code === 'BUDGET_HARDSTOP') throw e; // budget hard-stop stays fatal
|
|
253
|
-
if (!isTimeout(e)) { fatal = e; break; } // non-timeout errors stay fatal
|
|
390
|
+
if (!isTimeout(e)) { fatal = e; break; } // non-timeout errors stay fatal (a SUBSTRATE_MISMATCH among them)
|
|
254
391
|
if (budget) {
|
|
255
392
|
try {
|
|
256
393
|
if (e.phase === 'judge') budget.add((e.judgeAttempts || 1) * perCallCostUSD(judgeModel, 'judge'));
|
|
@@ -268,6 +405,9 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
268
405
|
samples: [],
|
|
269
406
|
mean: null,
|
|
270
407
|
stddev: null,
|
|
408
|
+
stop_reason: null,
|
|
409
|
+
truncated: false,
|
|
410
|
+
reported_model: null,
|
|
271
411
|
});
|
|
272
412
|
onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: String((e && e.message) || 'timeout') });
|
|
273
413
|
}
|
|
@@ -291,14 +431,22 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
291
431
|
// canary was dropped by exactly such an assembly silently gaining a field
|
|
292
432
|
// upstream that nothing here carried down (F-014-D).
|
|
293
433
|
variance_ratio_unavailable: agg.variance_ratio_unavailable,
|
|
434
|
+
// v0.6 (spec 026 AC-8): draws cut at the output cap, counted here so a
|
|
435
|
+
// reader sees it without walking the draw list.
|
|
436
|
+
n_truncated: draws.filter((d) => d.truncated === true).length,
|
|
294
437
|
draws,
|
|
295
438
|
};
|
|
296
439
|
|
|
297
|
-
// Every draw failed: the case is recorded
|
|
298
|
-
//
|
|
440
|
+
// Every draw failed: the case is recorded failed, carrying the draw list
|
|
441
|
+
// showing WHAT failed and how often. failed_timeout when every failure was
|
|
442
|
+
// a timeout; failed_unmeasured when any draw was unmeasured for another
|
|
443
|
+
// reason (spec 026 AC-3). The two share one predicate, caseFailed, in
|
|
444
|
+
// lib/receipt.js, and no reader excludes by either literal.
|
|
299
445
|
if (!last) {
|
|
300
446
|
failedCases += 1;
|
|
301
|
-
|
|
447
|
+
const status = nonTimeout ? FAILED_STATUSES[1] : FAILED_STATUSES[0];
|
|
448
|
+
const lastReason = [...draws].reverse().map((d) => d.reason).find(Boolean) || 'timeout';
|
|
449
|
+
return { caseResult: { id: c.id, mode, case_status: status, reason: lastReason, generation }, transcript: null };
|
|
302
450
|
}
|
|
303
451
|
|
|
304
452
|
// The v0.4-shaped fields now describe the DRAW SET, not one arbitrary draw,
|
|
@@ -317,7 +465,14 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
317
465
|
const caseResults = pairs.map((p) => p.caseResult);
|
|
318
466
|
const transcripts = keepTranscripts ? pairs.map((p) => p.transcript) : null;
|
|
319
467
|
|
|
320
|
-
|
|
468
|
+
// THE SURFACE IS WHAT ANSWERED, not what the runner would have chosen for
|
|
469
|
+
// the model id (spec 026 AC-1, F1). A stub run records surface stub, judge
|
|
470
|
+
// surface stub, answered_by.kind stub, and is UNVERIFIED: it measured nothing.
|
|
471
|
+
const answered = answeredByOf(replies, { attestedGen: attestedGen && replies.some((r) => r && r.answeredBy === 'model'), reportedAll });
|
|
472
|
+
const answeredBy = { kind: answered.kind, attested: answered.attested, reported_model: answered.attested ? modelId : null, reported_models: answered.reported_models, isolation: answered.isolation };
|
|
473
|
+
const surface = answered.kind === 'stub' ? 'stub' : surfaceForModel(modelId);
|
|
474
|
+
// v0.6 (spec 026 AC-11): which judge ran, and the template it graded with.
|
|
475
|
+
const judgeBlock = { ...judgeSettings(samples, judgeModel), ...(answered.kind === 'stub' ? { surface: 'stub' } : {}), model_id: judgeModel, prompt_template_hash: promptTemplateHash() };
|
|
321
476
|
const nowIso = opts.nowIso || new Date().toISOString();
|
|
322
477
|
// v0.4 economics. The pricing snapshot is frozen HERE, at run time, from the
|
|
323
478
|
// registry; every derived dollar figure below is computed from the snapshot and
|
|
@@ -359,24 +514,52 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
359
514
|
model_release_date: releaseDateFor(modelId),
|
|
360
515
|
provider: providerForModel(modelId),
|
|
361
516
|
surface,
|
|
362
|
-
// v0.3.1: on the openai/cli (codex) surface, record the fixed harness
|
|
517
|
+
// v0.3.1: on the openai/cli (codex) surface, record the fixed harness
|
|
518
|
+
// preamble. Absent on a stub run: it describes a harness that did not run.
|
|
363
519
|
surface_overhead_note: surface === 'openai-cli' ? CODEX_OVERHEAD_NOTE : undefined,
|
|
364
520
|
runner_version: RUNNER_VERSION,
|
|
365
521
|
date_utc: nowIso,
|
|
366
522
|
registry: registryStatus(modelId),
|
|
367
523
|
transcripts: keepTranscripts ? 'retained-local' : 'hashes-only',
|
|
368
|
-
judge:
|
|
524
|
+
judge: judgeBlock,
|
|
369
525
|
pricing_snapshot: pricingSnapshot,
|
|
526
|
+
answered_by: answeredBy,
|
|
370
527
|
},
|
|
371
528
|
cases: caseResults,
|
|
372
529
|
economics,
|
|
373
|
-
|
|
530
|
+
// The level is DERIVED from what answered: only a model-answered run may
|
|
531
|
+
// read TESTED. The schema refuses TESTED on a stub receipt as the second,
|
|
532
|
+
// independent control (spec 026 AC-1).
|
|
533
|
+
verificationLevel: answered.kind === 'model' ? 'TESTED' : 'UNVERIFIED',
|
|
374
534
|
});
|
|
375
535
|
|
|
376
536
|
return { receipt, calls, transcripts, failedCases };
|
|
377
537
|
}
|
|
378
538
|
|
|
379
|
-
|
|
539
|
+
// A band the formula could not form is printed as what it is, never as 0.000
|
|
540
|
+
// (spec 026 AC-7): one included case has a mean and no dispersion; none has
|
|
541
|
+
// neither.
|
|
542
|
+
function band(mean, sd) {
|
|
543
|
+
if (mean == null) return 'n/a (0 cases)';
|
|
544
|
+
if (sd == null) return `${mean.toFixed(3)} ± n/a (1 case)`;
|
|
545
|
+
return `${mean.toFixed(3)} ± ${sd.toFixed(3)}`;
|
|
546
|
+
}
|
|
547
|
+
// The comparison band, or the reason there is none.
|
|
548
|
+
function uncertaintyStr(cmp) {
|
|
549
|
+
if (cmp.delta_uncertainty != null) return cmp.delta_uncertainty.toFixed(3);
|
|
550
|
+
return `n/a (${cmp.delta_uncertainty_unavailable === 'single_case' ? '1 case' : cmp.delta_uncertainty_unavailable === 'no_cases' ? '0 cases' : 'no band'})`;
|
|
551
|
+
}
|
|
552
|
+
// The answered_by block, said in one line for the summary and the CLI.
|
|
553
|
+
function answeredLine(receipt) {
|
|
554
|
+
const ab = (receipt.run && receipt.run.answered_by) || null;
|
|
555
|
+
if (!ab) return 'answered by: unrecorded (pre-v0.6 receipt)';
|
|
556
|
+
if (ab.kind === 'stub') return 'answered by: stub (DRIFTPROOF_STUB) — nothing answered; this run measured nothing (UNVERIFIED)';
|
|
557
|
+
if (ab.kind === 'external') return 'answered by: an external tool (imported)';
|
|
558
|
+
const iso = ab.isolation === 'eval-user' ? 'the isolated eval-user hop' : ab.isolation === 'same-user' ? 'the same-user spawn (--trusted-skill)' : 'no spawn (api surface)';
|
|
559
|
+
return ab.attested
|
|
560
|
+
? `answered by: model ${ab.reported_model} (attested by the surface; ${iso})`
|
|
561
|
+
: `answered by: model, but the surface did not report which model answered (attested: false; ${iso})`;
|
|
562
|
+
}
|
|
380
563
|
|
|
381
564
|
// Render a short human-readable markdown summary of a receipt.
|
|
382
565
|
function summarizeReceipt(receipt) {
|
|
@@ -385,6 +568,7 @@ function summarizeReceipt(receipt) {
|
|
|
385
568
|
L.push('');
|
|
386
569
|
L.push(`- **model:** \`${receipt.run.model_id}\`${receipt.run.model_release_date ? ` (released ${receipt.run.model_release_date})` : ''}`);
|
|
387
570
|
L.push(`- **surface:** ${receipt.run.surface}`);
|
|
571
|
+
L.push(`- **${answeredLine(receipt)}**`);
|
|
388
572
|
L.push(`- **run (UTC):** ${receipt.run.date_utc}`);
|
|
389
573
|
L.push(`- **runner:** v${receipt.run.runner_version}`);
|
|
390
574
|
const j = receipt.run.judge || {};
|
|
@@ -399,22 +583,39 @@ function summarizeReceipt(receipt) {
|
|
|
399
583
|
L.push('');
|
|
400
584
|
const cmp = receipt.comparison;
|
|
401
585
|
const aggs = receipt.results.aggregates;
|
|
402
|
-
|
|
586
|
+
if ((receipt.run.answered_by || {}).kind === 'stub') {
|
|
587
|
+
L.push('> **STUB RUN** — DRIFTPROOF_STUB=1: nothing answered, the text was canned, and this run measured nothing. The receipt is UNVERIFIED and verdicts nothing.');
|
|
588
|
+
L.push('');
|
|
589
|
+
}
|
|
403
590
|
if (receipt.run.status === 'incomplete') {
|
|
404
591
|
L.push(`> ⏱ **INCOMPLETE** — ${receipt.run.failed_case_count} case(s) had an arm that could not be measured and are EXCLUDED from the aggregates below, BOTH arms together; this receipt must not be used to compute a drift/durability verdict.`);
|
|
405
592
|
L.push('');
|
|
406
593
|
}
|
|
594
|
+
// Spec 026 AC-8: draws cut at the output cap, said once for the run.
|
|
595
|
+
const nTruncated = receipt.results.cases.reduce((a, c) => a + (((c.generation || {}).n_truncated) || 0), 0);
|
|
596
|
+
if (nTruncated) {
|
|
597
|
+
L.push(`> ✂ **${nTruncated} draw(s) truncated** at the output cap: each is unmeasured and excluded from every band below.`);
|
|
598
|
+
L.push('');
|
|
599
|
+
}
|
|
407
600
|
L.push(`with_skill **${band(aggs.with_skill.mean_score, aggs.with_skill.stddev)}** vs baseline **${band(aggs.baseline.mean_score, aggs.baseline.stddev)}**`);
|
|
408
601
|
L.push('');
|
|
409
|
-
|
|
602
|
+
if (cmp.delta == null) {
|
|
603
|
+
L.push(`skill lift **n/a** (${cmp.delta_uncertainty_unavailable === 'no_cases' ? 'no case was included on an arm' : 'no comparison'})`);
|
|
604
|
+
} else {
|
|
605
|
+
const sign = cmp.delta >= 0 ? '+' : '';
|
|
606
|
+
L.push(`skill lift **${sign}${cmp.delta.toFixed(3)}** (combined uncertainty ± ${uncertaintyStr(cmp)})`);
|
|
607
|
+
}
|
|
608
|
+
L.push('');
|
|
609
|
+
// The rule the bands above are derived by, said beside them (spec 026 AC-6).
|
|
610
|
+
L.push(`band rule: each arm's band is the sample stddev of its per-case means${aggs.band_rule ? ` — ${aggs.band_rule}` : ' (unstated on this pre-v0.6 receipt)'}`);
|
|
410
611
|
L.push('');
|
|
411
612
|
L.push(`## Per-case (mean ± stddev over ${(receipt.run.judge || {}).samples || 1} judge samples)`);
|
|
412
613
|
L.push('');
|
|
413
614
|
L.push(`| case | mode | outcome | mean ± stddev | judge reason |`);
|
|
414
615
|
L.push(`|---|---|---|---|---|`);
|
|
415
616
|
for (const c of receipt.results.cases) {
|
|
416
|
-
if (c
|
|
417
|
-
L.push(`| \`${c.id}\` | ${c.mode} | ⏱
|
|
617
|
+
if (caseFailed(c)) {
|
|
618
|
+
L.push(`| \`${c.id}\` | ${c.mode} | ⏱ ${c.case_status} | — (not measured) | ${c.reason || 'not measured'} |`);
|
|
418
619
|
continue;
|
|
419
620
|
}
|
|
420
621
|
const flag = c.outcome === 'borderline' ? ' ⚠' : '';
|
|
@@ -424,4 +625,4 @@ function summarizeReceipt(receipt) {
|
|
|
424
625
|
return L.join('\n');
|
|
425
626
|
}
|
|
426
627
|
|
|
427
|
-
module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor, resolveCallTimeoutMs };
|
|
628
|
+
module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor, resolveCallTimeoutMs, answeredLine, answeredByOf, attest, band, uncertaintyStr };
|
package/lib/runner.js
CHANGED
|
@@ -39,13 +39,38 @@ class Gate {
|
|
|
39
39
|
return rec.pass;
|
|
40
40
|
}
|
|
41
41
|
|
|
42
|
+
// Record an assertion whose SUBJECT this checkout does not carry (spec 030
|
|
43
|
+
// AC-5). Not a pass and not a failure: a row that says the check could not
|
|
44
|
+
// apply here and why.
|
|
45
|
+
//
|
|
46
|
+
// This exists because the two wrong answers are both worse. Guarding such a
|
|
47
|
+
// check with `if (...)` REMOVES the row, and the headline count silently
|
|
48
|
+
// drops by one - the register's `absence-vs-unreadable` class, and the reason
|
|
49
|
+
// the DECISIONS #4 check was written to fail rather than skip. Failing it
|
|
50
|
+
// instead means a `--depth 1` clone, which carries no `main` ref and so has
|
|
51
|
+
// no merge range to read, can never reach a green gate however correct it is.
|
|
52
|
+
// A third state costs one row and keeps both properties: the assertion always
|
|
53
|
+
// registers, and its cause is printed rather than hidden in a detail field
|
|
54
|
+
// that only prints on failure.
|
|
55
|
+
//
|
|
56
|
+
// `notApplicable` is NEVER the answer to "I could not read it". An absence
|
|
57
|
+
// must be positively detected - the ref is not there - or it is a failure.
|
|
58
|
+
notApplicable(name, cause) {
|
|
59
|
+
const rec = { section: this._section, name, pass: true, notApplicable: true, cause, detail: null };
|
|
60
|
+
this.results.push(rec);
|
|
61
|
+
// eslint-disable-next-line no-console
|
|
62
|
+
console.log(` [N/A] ${name} -- ${cause}`);
|
|
63
|
+
return true;
|
|
64
|
+
}
|
|
65
|
+
|
|
42
66
|
// Convenience: assert deep equality of two JSON-able values.
|
|
43
67
|
checkEqual(name, actual, expected) {
|
|
44
68
|
const pass = safeJson(actual) === safeJson(expected);
|
|
45
69
|
return this.check(name, pass, { actual, expected });
|
|
46
70
|
}
|
|
47
71
|
|
|
48
|
-
get passed() { return this.results.filter((r) => r.pass).length; }
|
|
72
|
+
get passed() { return this.results.filter((r) => r.pass && !r.notApplicable).length; }
|
|
73
|
+
get notApplicableCount() { return this.results.filter((r) => r.notApplicable).length; }
|
|
49
74
|
get failed() { return this.results.filter((r) => !r.pass); }
|
|
50
75
|
get total() { return this.results.length; }
|
|
51
76
|
|
|
@@ -53,12 +78,19 @@ class Gate {
|
|
|
53
78
|
summarize() {
|
|
54
79
|
const failed = this.failed;
|
|
55
80
|
// eslint-disable-next-line no-console
|
|
56
|
-
|
|
81
|
+
const na = this.notApplicableCount;
|
|
82
|
+
const naSuffix = na ? `, ${na} not applicable` : '';
|
|
83
|
+
// eslint-disable-next-line no-console
|
|
84
|
+
console.log(`\n=== ${this.title.toUpperCase()} RESULT: ${this.passed}/${this.total - na} passed, ${failed.length} failed${naSuffix} ===`);
|
|
85
|
+
for (const r of this.results.filter((x) => x.notApplicable)) {
|
|
86
|
+
// eslint-disable-next-line no-console
|
|
87
|
+
console.log(` N/A [${r.section}] ${r.name}: ${r.cause}`);
|
|
88
|
+
}
|
|
57
89
|
for (const f of failed) {
|
|
58
90
|
// eslint-disable-next-line no-console
|
|
59
91
|
console.log(` FAIL [${f.section}] ${f.name}: ${safeJson(f.detail)}`);
|
|
60
92
|
}
|
|
61
|
-
return { title: this.title, total: this.total, passed: this.passed, failed: failed.length, results: this.results };
|
|
93
|
+
return { title: this.title, total: this.total, passed: this.passed, failed: failed.length, notApplicable: na, results: this.results };
|
|
62
94
|
}
|
|
63
95
|
|
|
64
96
|
toExitCode() {
|
package/lib/skill.js
CHANGED
|
@@ -4,7 +4,16 @@
|
|
|
4
4
|
const fs = require('fs');
|
|
5
5
|
const path = require('path');
|
|
6
6
|
const { sha256Files, sha256Canonical } = require('./canonical');
|
|
7
|
-
const { SUITE_FORMAT } = require('../config');
|
|
7
|
+
const { SUITE_FORMAT, SKILL_MAX_FILES, SKILL_MAX_BYTES, SKILL_MAX_DEPTH, SUITE_MAX_CASES, CASE_MAX_CHARS } = require('../config');
|
|
8
|
+
|
|
9
|
+
// A bound was exceeded: the error names the bound and the value (spec 026
|
|
10
|
+
// AC-13), so the refusal says what to change rather than that something is
|
|
11
|
+
// too big.
|
|
12
|
+
function pastBound(bound, limit, value, what) {
|
|
13
|
+
const e = new Error(`${what}: ${value} exceeds ${bound} (${limit}); refusing to load`);
|
|
14
|
+
e.code = 'INPUT_BOUND'; e.bound = bound; e.limit = limit; e.value = value;
|
|
15
|
+
return e;
|
|
16
|
+
}
|
|
8
17
|
|
|
9
18
|
// Load a skill directory and its eval suite.
|
|
10
19
|
//
|
|
@@ -21,13 +30,22 @@ const { SUITE_FORMAT } = require('../config');
|
|
|
21
30
|
const IGNORE_DIRS = new Set(['.git', 'node_modules', 'evals']);
|
|
22
31
|
const IGNORE_FILES = new Set(['.DS_Store']);
|
|
23
32
|
|
|
24
|
-
|
|
33
|
+
// Bounded (spec 026 AC-13): the walk stops at SKILL_MAX_DEPTH directories
|
|
34
|
+
// below the skill dir, SKILL_MAX_FILES bundled files, and SKILL_MAX_BYTES of
|
|
35
|
+
// them together, and throws naming the bound the moment one is passed, so a
|
|
36
|
+
// pathological tree is refused before its bytes are read into memory.
|
|
37
|
+
function walkFiles(dir, base = dir, acc = [], state = { bytes: 0 }, depth = 0) {
|
|
38
|
+
if (depth > SKILL_MAX_DEPTH) throw pastBound('SKILL_MAX_DEPTH', SKILL_MAX_DEPTH, depth, `directory depth under ${base}`);
|
|
25
39
|
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
|
26
40
|
if (entry.isDirectory()) {
|
|
27
41
|
if (IGNORE_DIRS.has(entry.name)) continue;
|
|
28
|
-
walkFiles(path.join(dir, entry.name), base, acc);
|
|
42
|
+
walkFiles(path.join(dir, entry.name), base, acc, state, depth + 1);
|
|
29
43
|
} else if (entry.isFile() && !IGNORE_FILES.has(entry.name)) {
|
|
30
44
|
const abs = path.join(dir, entry.name);
|
|
45
|
+
if (acc.length + 1 > SKILL_MAX_FILES) throw pastBound('SKILL_MAX_FILES', SKILL_MAX_FILES, acc.length + 1, `bundled files under ${base}`);
|
|
46
|
+
const size = fs.statSync(abs).size;
|
|
47
|
+
state.bytes += size;
|
|
48
|
+
if (state.bytes > SKILL_MAX_BYTES) throw pastBound('SKILL_MAX_BYTES', SKILL_MAX_BYTES, state.bytes, `bundled bytes under ${base}`);
|
|
31
49
|
acc.push({ path: path.relative(base, abs), bytes: fs.readFileSync(abs) });
|
|
32
50
|
}
|
|
33
51
|
}
|
|
@@ -108,12 +126,16 @@ function normalizeCases(raw) {
|
|
|
108
126
|
: Array.isArray(raw.evals) ? raw.evals
|
|
109
127
|
: null;
|
|
110
128
|
if (!list) throw new Error('evals.json must be an array or have a `cases`/`evals` array');
|
|
129
|
+
// Bounded (spec 026 AC-13): the case count, and each prompt and rubric.
|
|
130
|
+
if (list.length > SUITE_MAX_CASES) throw pastBound('SUITE_MAX_CASES', SUITE_MAX_CASES, list.length, 'cases in the suite');
|
|
111
131
|
return list.map((c, i) => {
|
|
112
132
|
const id = String(c.id || c.name || `case-${i + 1}`);
|
|
113
133
|
const prompt = c.prompt || c.input || c.task;
|
|
114
134
|
const rubric = c.rubric || c.criteria || c.expected;
|
|
115
135
|
if (!prompt) throw new Error(`case "${id}" is missing a prompt/input/task`);
|
|
116
136
|
if (!rubric) throw new Error(`case "${id}" is missing a rubric/criteria/expected`);
|
|
137
|
+
if (String(prompt).length > CASE_MAX_CHARS) throw pastBound('CASE_MAX_CHARS', CASE_MAX_CHARS, String(prompt).length, `case "${id}" prompt characters`);
|
|
138
|
+
if (String(rubric).length > CASE_MAX_CHARS) throw pastBound('CASE_MAX_CHARS', CASE_MAX_CHARS, String(rubric).length, `case "${id}" rubric characters`);
|
|
117
139
|
const threshold = typeof c.pass_threshold === 'number' ? c.pass_threshold
|
|
118
140
|
: typeof c.threshold === 'number' ? c.threshold : 0.7;
|
|
119
141
|
const norm = { id, prompt: String(prompt), rubric: String(rubric), pass_threshold: threshold };
|
|
@@ -124,4 +146,4 @@ function normalizeCases(raw) {
|
|
|
124
146
|
});
|
|
125
147
|
}
|
|
126
148
|
|
|
127
|
-
module.exports = { loadSkill, normalizeCases, parseSkillMeta };
|
|
149
|
+
module.exports = { loadSkill, normalizeCases, parseSkillMeta, pastBound };
|
package/lib/stats.js
CHANGED
|
@@ -31,8 +31,11 @@ function stderr(xs) {
|
|
|
31
31
|
return round(stddev(xs) / Math.sqrt(n));
|
|
32
32
|
}
|
|
33
33
|
|
|
34
|
-
// Combine independent uncertainties in quadrature: sqrt(a^2 + b^2).
|
|
34
|
+
// Combine independent uncertainties in quadrature: sqrt(a^2 + b^2). Null when
|
|
35
|
+
// either band is null: a combination of a band that could not form cannot form
|
|
36
|
+
// either (spec 026 AC-7), and the receipt says why beside it.
|
|
35
37
|
function combineUncertainty(a, b) {
|
|
38
|
+
if (a == null || b == null) return null;
|
|
36
39
|
return round(Math.sqrt(a * a + b * b));
|
|
37
40
|
}
|
|
38
41
|
|
|
@@ -45,10 +48,16 @@ function combineUncertainty(a, b) {
|
|
|
45
48
|
// is a conventional, honest "mean ± stddev across the suite". The drift HEADLINE
|
|
46
49
|
// verdict is driven by the per-case band-overlap verdicts (see lib/diff.js), not
|
|
47
50
|
// by this aggregate band; this value is a reported summary statistic.
|
|
51
|
+
//
|
|
52
|
+
// A BAND THE FORMULA CANNOT FORM IS NULL, NEVER 0 (spec 026 AC-7, F3). The
|
|
53
|
+
// sample standard deviation of one value is undefined, and the mean of no
|
|
54
|
+
// values is not a number; printing 0.000 for either asserted a precision that
|
|
55
|
+
// was never measured (the one-case run's `± 0.000`). The rule the receipt
|
|
56
|
+
// states (results.aggregates.band_rule) is this function.
|
|
48
57
|
function aggregateBands(cases) {
|
|
49
|
-
if (!cases.length) return { mean:
|
|
58
|
+
if (!cases.length) return { mean: null, stddev: null };
|
|
50
59
|
const means = cases.map((c) => c.mean);
|
|
51
|
-
return { mean: mean(means), stddev: stddev(means) };
|
|
60
|
+
return { mean: mean(means), stddev: means.length < 2 ? null : stddev(means) };
|
|
52
61
|
}
|
|
53
62
|
|
|
54
63
|
// Do two confidence bands (mean ± half-width) fail to overlap, and in which
|