driftproof 0.8.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +45 -5
- package/bin/driftproof +191 -22
- package/config.js +26 -2
- package/lib/checks.js +28 -6
- package/lib/diff.js +58 -4
- package/lib/importers.js +8 -3
- package/lib/judge.js +74 -15
- package/lib/models.js +52 -11
- package/lib/provider.js +265 -103
- package/lib/receipt.js +51 -11
- package/lib/run.js +231 -26
- package/lib/skill.js +26 -4
- package/lib/stats.js +12 -3
- package/lib/usage.js +56 -4
- package/lib/value.js +4 -2
- package/lib/verdict.js +38 -7
- package/package.json +3 -2
- package/spec/RECEIPT.md +68 -11
- package/spec/receipt.schema.json +299 -54
- package/spec/receipt.v0.5.schema.json +1318 -0
package/lib/run.js
CHANGED
|
@@ -2,16 +2,18 @@
|
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
4
|
const { complete, resolveModel, surfaceForModel, isMeteredSurface, retryPolicyForSurface, CODEX_OVERHEAD_NOTE } = require('./provider');
|
|
5
|
-
const { gradeSamples, judgeSettings } = require('./judge');
|
|
6
|
-
const { buildReceipt } = require('./receipt');
|
|
5
|
+
const { gradeSamples, judgeSettings, promptTemplateHash, isTruncated } = require('./judge');
|
|
6
|
+
const { buildReceipt, caseFailed, FAILED_STATUSES } = require('./receipt');
|
|
7
7
|
const { sha256 } = require('./canonical');
|
|
8
|
-
const { registryStatus, providerForModel, priceForModel } = require('./models');
|
|
8
|
+
const { registryStatus, providerForModel, priceForModel, assertRegistered } = require('./models');
|
|
9
9
|
const { perCallCostUSD } = require('./cost');
|
|
10
10
|
const { runChecks } = require('./checks');
|
|
11
11
|
const { estimateTokens } = require('./skillCost');
|
|
12
12
|
const { hasUsage, normalizeUsage } = require('./usage');
|
|
13
13
|
const { buildPricingSnapshot, computeEconomics } = require('./value');
|
|
14
14
|
const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES, DEV_MAX_CALLS } = require('../config');
|
|
15
|
+
const { stubEnabled } = require('./stub');
|
|
16
|
+
const { canonicalModelId } = require('./usage');
|
|
15
17
|
const { SAMPLING, acrossDraws, nextAction } = require('./sampling');
|
|
16
18
|
const { suiteCanary } = require('./canary');
|
|
17
19
|
|
|
@@ -51,7 +53,7 @@ function projectCalls(caseCount, samples, draws = 1) {
|
|
|
51
53
|
// Ask the target model to perform one eval case. `withSkill` decides whether the
|
|
52
54
|
// SKILL.md is prepended as a system prompt (the whole point: measure the skill's
|
|
53
55
|
// marginal effect vs a bare baseline).
|
|
54
|
-
async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs }) {
|
|
56
|
+
async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs, trusted = false }) {
|
|
55
57
|
// Test seam (gate only): force a persistent timeout for a named case id so the
|
|
56
58
|
// failed_timeout path is exercised deterministically without any live call.
|
|
57
59
|
if (process.env.DRIFTPROOF_TEST_TIMEOUT_CASEID && process.env.DRIFTPROOF_TEST_TIMEOUT_CASEID === caseObj.id) {
|
|
@@ -60,8 +62,44 @@ async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs }) {
|
|
|
60
62
|
throw e;
|
|
61
63
|
}
|
|
62
64
|
const system = withSkill ? skillMd : undefined;
|
|
63
|
-
const
|
|
64
|
-
|
|
65
|
+
const out = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs, trusted });
|
|
66
|
+
// The whole reply travels: what answered, what it said served the call, why
|
|
67
|
+
// it stopped, and which spawn path was taken (spec 026 AC-1, AC-2, AC-8).
|
|
68
|
+
return { text: out.text, usage: out.usage, wall_ms: out.wall_ms, attempts: out.attempts, answeredBy: out.answeredBy, surface: out.surface, reportedModels: out.reportedModels, stopReason: out.stopReason, isolation: out.isolation };
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
// ── what answered (spec 026, AC-2) ───────────────────────────────────────────
|
|
72
|
+
// The surface's echo is compared with the requested id on CANONICAL ids (the
|
|
73
|
+
// real claude CLI keys its usage by the undated form). A surface that names a
|
|
74
|
+
// DIFFERENT model stops the run before the next call; no receipt is written,
|
|
75
|
+
// because a receipt naming a model that did not answer is the defect this
|
|
76
|
+
// exists to close. A surface that names nothing is recorded as attested: false.
|
|
77
|
+
function attest(reply, requestedId, phase) {
|
|
78
|
+
const reported = Array.isArray(reply && reply.reportedModels) ? reply.reportedModels : null;
|
|
79
|
+
if (!reported || !reported.length) return { attested: false, reported };
|
|
80
|
+
const want = canonicalModelId(resolveModel(requestedId));
|
|
81
|
+
const other = reported.find((id) => canonicalModelId(id) !== want);
|
|
82
|
+
if (other) {
|
|
83
|
+
const e = new Error(`substrate mismatch: the ${phase} surface answered as "${other}" where "${requestedId}" (canonical "${want}") was requested; the run stops before the next call and no receipt is written`);
|
|
84
|
+
e.code = 'SUBSTRATE_MISMATCH';
|
|
85
|
+
e.requested = requestedId; e.reported = other;
|
|
86
|
+
throw e;
|
|
87
|
+
}
|
|
88
|
+
return { attested: true, reported };
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
// The run's answered_by block, derived from every reply the run received and
|
|
92
|
+
// never from the surface the runner would have chosen (spec 026 AC-1: a stub
|
|
93
|
+
// run used to record the real surface name because the name was computed
|
|
94
|
+
// from the model id, not from what answered). With no reply at all (every
|
|
95
|
+
// call failed before answering) the kind is what the process would have
|
|
96
|
+
// answered with, which is the one thing still known.
|
|
97
|
+
function answeredByOf(replies, { attestedGen = false, reportedAll = new Set() } = {}) {
|
|
98
|
+
const kinds = new Set(replies.map((r) => r && r.answeredBy).filter(Boolean));
|
|
99
|
+
const kind = replies.length ? (kinds.size === 1 && kinds.has('stub') ? 'stub' : 'model') : (stubEnabled() ? 'stub' : 'model');
|
|
100
|
+
const iso = replies.map((r) => r && r.isolation).find(Boolean) || 'none';
|
|
101
|
+
const reportedModels = reportedAll.size ? [...reportedAll].sort() : null;
|
|
102
|
+
return { kind, attested: kind === 'model' && attestedGen, reported_models: reportedModels, isolation: iso };
|
|
65
103
|
}
|
|
66
104
|
|
|
67
105
|
// Determine a case outcome from its sampled band and threshold.
|
|
@@ -78,8 +116,14 @@ function outcomeFor(mean, stddev, threshold) {
|
|
|
78
116
|
// `generationHash` binds the graded case to the exact generation text (v0.3).
|
|
79
117
|
// Returns { caseResult, sampleTexts } — sampleTexts is transient (retained only
|
|
80
118
|
// under --keep-transcripts; never part of the receipt).
|
|
81
|
-
async function judgeCase({ caseObj, response, generationHash, judgeModel, mode, timeoutMs, samples }) {
|
|
82
|
-
const g = await gradeSamples({ task: caseObj.prompt, response, rubric: caseObj.rubric, model: judgeModel, samples, timeoutMs });
|
|
119
|
+
async function judgeCase({ caseObj, response, generationHash, judgeModel, mode, timeoutMs, samples, trusted = false }) {
|
|
120
|
+
const g = await gradeSamples({ task: caseObj.prompt, response, rubric: caseObj.rubric, model: judgeModel, samples, timeoutMs, trusted });
|
|
121
|
+
// A judge that produced no score produced no measurement (spec 026 AC-3):
|
|
122
|
+
// the draw is unmeasured with the judge's reason, and no sample is kept.
|
|
123
|
+
// The judge calls that WERE made travel with the unmeasured result: their
|
|
124
|
+
// output hashes (transcript auditability: the judge said something, and a
|
|
125
|
+
// reader can check what) and their usage, so the receipt records every call.
|
|
126
|
+
if (g.unmeasured) return { unmeasured: true, reason: g.reason, sampleTexts: g.sample_texts, sampleHashes: g.sample_hashes || [], attempts: g.attempts, replies: g.replies || [], judge_usage: hasUsage(g.usage) ? g.usage : null };
|
|
83
127
|
const outcome = outcomeFor(g.mean, g.stddev, caseObj.pass_threshold);
|
|
84
128
|
const caseResult = {
|
|
85
129
|
id: caseObj.id,
|
|
@@ -102,7 +146,7 @@ async function judgeCase({ caseObj, response, generationHash, judgeModel, mode,
|
|
|
102
146
|
if (checks.length) caseResult.checks = checks;
|
|
103
147
|
// v0.4: grading overhead for this case row, kept OUT of the skill-value math.
|
|
104
148
|
if (hasUsage(g.usage)) caseResult.judge_usage = g.usage;
|
|
105
|
-
return { caseResult, sampleTexts: g.sample_texts, attempts: g.attempts };
|
|
149
|
+
return { caseResult, sampleTexts: g.sample_texts, attempts: g.attempts, replies: g.replies || [] };
|
|
106
150
|
}
|
|
107
151
|
|
|
108
152
|
// Run up to `concurrency` async tasks at a time, preserving input order in the
|
|
@@ -127,7 +171,7 @@ async function mapPool(items, concurrency, fn) {
|
|
|
127
171
|
// a sealed receipt. Enforces a hard call cap; every model+judge call counts.
|
|
128
172
|
//
|
|
129
173
|
// opts: { maxCases, maxCalls, samples, judgeModel, timeoutMs, concurrency,
|
|
130
|
-
// onProgress, budget, keepTranscripts, nowIso }
|
|
174
|
+
// onProgress, budget, keepTranscripts, nowIso, trusted }
|
|
131
175
|
// budget — optional BudgetTracker; accumulates estimated per-call
|
|
132
176
|
// USD as the run proceeds and hard-stops at 1.25× the cap.
|
|
133
177
|
// keepTranscripts — when true, the run records transcripts:"retained-local"
|
|
@@ -155,6 +199,13 @@ const resolveCallTimeoutMs = function resolveCallTimeoutMs(surface, opts = {}) {
|
|
|
155
199
|
async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
156
200
|
const modelId = resolveModel(model);
|
|
157
201
|
const judgeModel = opts.judgeModel ? resolveModel(opts.judgeModel) : modelId;
|
|
202
|
+
// Spec 026 AC-10 (F6): the target and the judge must be models the registry
|
|
203
|
+
// knows, checked HERE, on the path the CLI, the report scripts and the
|
|
204
|
+
// trigger all share, before the cost guard and before any call. bin/driftproof
|
|
205
|
+
// makes the same check at its door so the refusal names the registry path in
|
|
206
|
+
// its own message; this one is the door every caller passes.
|
|
207
|
+
assertRegistered(modelId, 'model');
|
|
208
|
+
assertRegistered(judgeModel, 'judge model');
|
|
158
209
|
// THE SURFACE'S OWN DECLARED TIMEOUT, resolved per arm (spec 017 AC-1).
|
|
159
210
|
//
|
|
160
211
|
// This read `opts.timeoutMs || 120000`, and `lib/provider.js` documents that an
|
|
@@ -177,6 +228,10 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
177
228
|
const onProgress = opts.onProgress || (() => {});
|
|
178
229
|
const budget = opts.budget || null;
|
|
179
230
|
const keepTranscripts = !!opts.keepTranscripts;
|
|
231
|
+
// spec 022: the SAME-USER legacy spawn is reachable only when a caller says
|
|
232
|
+
// `trusted: true` (bin/driftproof --trusted-skill). Every other caller of this
|
|
233
|
+
// function, the report scripts and the release watcher included, isolates.
|
|
234
|
+
const trusted = !!opts.trusted;
|
|
180
235
|
|
|
181
236
|
let cases = skill.suite.cases;
|
|
182
237
|
if (opts.maxCases && cases.length > opts.maxCases) cases = cases.slice(0, opts.maxCases);
|
|
@@ -200,6 +255,12 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
200
255
|
|
|
201
256
|
let calls = 0;
|
|
202
257
|
let failedCases = 0;
|
|
258
|
+
// Spec 026 AC-1, AC-2: every reply the run received, for the answered_by
|
|
259
|
+
// block; whether every generation reply attested the requested model (read
|
|
260
|
+
// from the surface's echo, AC-2); every canonical id any reply named.
|
|
261
|
+
const replies = [];
|
|
262
|
+
let attestedGen = true;
|
|
263
|
+
const reportedAll = new Set();
|
|
203
264
|
const isTimeout = (e) => !!(e && (e.code === 'TIMEOUT' || /tim(e|ed)\s*out/i.test(String((e && e.message) || ''))));
|
|
204
265
|
const genKind = (ws) => (ws ? 'gen_with_skill' : 'gen_baseline');
|
|
205
266
|
const pairs = await mapPool(tasks, concurrency, async ({ c, withSkill }) => {
|
|
@@ -217,19 +278,93 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
217
278
|
let lastTranscript = null;
|
|
218
279
|
let action = { stop: false, reason: 'below_min' };
|
|
219
280
|
let fatal = null;
|
|
281
|
+
// Spec 026 AC-3: an unmeasured draw that was NOT a timeout (an empty
|
|
282
|
+
// generation, a judge with no score) makes the case failed_unmeasured
|
|
283
|
+
// rather than failed_timeout when no draw measured.
|
|
284
|
+
let nonTimeout = false;
|
|
285
|
+
|
|
286
|
+
// An UNMEASURED draw carries no score, no fabricated samples, and the
|
|
287
|
+
// reason it carries none; it is excluded from every statistic rather than
|
|
288
|
+
// counted as a zero. v0.6 adds what the surface said about the reply. The
|
|
289
|
+
// generation's own hash is kept when there was text.
|
|
290
|
+
const unmeasuredDraw = (drawIndex, reason, gen) => ({
|
|
291
|
+
draw_index: drawIndex,
|
|
292
|
+
generation_hash: gen && String(gen.text || '') ? sha256(String(gen.text || '')) : null,
|
|
293
|
+
status: 'unmeasured',
|
|
294
|
+
reason: String(reason || '').slice(0, 200),
|
|
295
|
+
samples: [],
|
|
296
|
+
mean: null,
|
|
297
|
+
stddev: null,
|
|
298
|
+
stop_reason: gen ? (gen.stopReason || null) : null,
|
|
299
|
+
truncated: !!(gen && gen.truncated === true),
|
|
300
|
+
reported_model: gen && gen.attested ? modelId : null,
|
|
301
|
+
});
|
|
220
302
|
|
|
221
303
|
while (!action.stop && draws.length < SAMPLING.max) {
|
|
222
304
|
const drawIndex = draws.length;
|
|
223
305
|
try {
|
|
224
306
|
onProgress({ case: c.id, mode, phase: 'generate', draw: drawIndex });
|
|
225
|
-
const gen = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs: ctGen });
|
|
307
|
+
const gen = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs: ctGen, trusted });
|
|
226
308
|
calls += 1;
|
|
227
309
|
if (budget) budget.add((gen.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
|
|
310
|
+
replies.push(gen);
|
|
311
|
+
// What answered, on canonical ids; a different model stops the run.
|
|
312
|
+
const a = attest(gen, modelId, 'generation');
|
|
313
|
+
gen.attested = a.attested;
|
|
314
|
+
if (!a.attested) attestedGen = false;
|
|
315
|
+
for (const id of (a.reported || [])) reportedAll.add(canonicalModelId(id));
|
|
316
|
+
gen.truncated = isTruncated(gen.stopReason);
|
|
317
|
+
// Spec 026 AC-8 (F4): a generation cut at the output cap is a partial
|
|
318
|
+
// answer, and a partial answer graded as a whole one is a score about
|
|
319
|
+
// something the model did not write. Recorded truncated, unmeasured,
|
|
320
|
+
// never judged.
|
|
321
|
+
if (gen.truncated === true) {
|
|
322
|
+
nonTimeout = true;
|
|
323
|
+
const d = unmeasuredDraw(drawIndex, `generation truncated at the output cap (stop_reason ${gen.stopReason})`, gen);
|
|
324
|
+
if (hasUsage(gen.usage)) d.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
|
|
325
|
+
draws.push(d);
|
|
326
|
+
onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: d.reason });
|
|
327
|
+
action = nextAction(draws);
|
|
328
|
+
continue;
|
|
329
|
+
}
|
|
330
|
+
// Spec 026 AC-3 (F2): an empty generation is the absence of an output,
|
|
331
|
+
// not an output that scored zero. It is not sent to the judge.
|
|
332
|
+
if (!String(gen.text || '').trim()) {
|
|
333
|
+
nonTimeout = true;
|
|
334
|
+
const d = unmeasuredDraw(drawIndex, 'empty generation (the surface returned no text)', gen);
|
|
335
|
+
if (hasUsage(gen.usage)) d.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
|
|
336
|
+
draws.push(d);
|
|
337
|
+
onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: d.reason });
|
|
338
|
+
action = nextAction(draws);
|
|
339
|
+
continue;
|
|
340
|
+
}
|
|
228
341
|
const generationHash = sha256(String(gen.text || ''));
|
|
229
342
|
onProgress({ case: c.id, mode, phase: 'judge', samples, draw: drawIndex });
|
|
230
|
-
const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ctJudge, samples });
|
|
343
|
+
const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ctJudge, samples, trusted });
|
|
231
344
|
calls += samples;
|
|
232
345
|
if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judgeModel, 'judge'));
|
|
346
|
+
for (const r of (jr.replies || [])) {
|
|
347
|
+
if (!r) continue;
|
|
348
|
+
replies.push(r);
|
|
349
|
+
const ja = attest(r, judgeModel, 'judge');
|
|
350
|
+
for (const id of (ja.reported || [])) reportedAll.add(canonicalModelId(id));
|
|
351
|
+
}
|
|
352
|
+
if (jr.unmeasured) {
|
|
353
|
+
// The judge returned no score for this draw (empty, unparseable,
|
|
354
|
+
// non-numeric, out of range): unmeasured, naming the judge's reason;
|
|
355
|
+
// the generation's own hash is kept.
|
|
356
|
+
nonTimeout = true;
|
|
357
|
+
const d = unmeasuredDraw(drawIndex, jr.reason, gen);
|
|
358
|
+
// The judge samples taken before the draw was called unmeasured are
|
|
359
|
+
// recorded by hash (no score entered any statistic; the calls happened).
|
|
360
|
+
if ((jr.sampleHashes || []).length) d.judge_sample_hashes = jr.sampleHashes;
|
|
361
|
+
if (hasUsage(gen.usage)) d.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
|
|
362
|
+
if (jr.judge_usage) d.judge_usage = jr.judge_usage;
|
|
363
|
+
draws.push(d);
|
|
364
|
+
onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: d.reason });
|
|
365
|
+
action = nextAction(draws);
|
|
366
|
+
continue;
|
|
367
|
+
}
|
|
233
368
|
const draw = {
|
|
234
369
|
draw_index: drawIndex,
|
|
235
370
|
generation_hash: generationHash,
|
|
@@ -238,6 +373,12 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
238
373
|
judge_sample_hashes: jr.caseResult.judge_sample_hashes,
|
|
239
374
|
mean: jr.caseResult.mean,
|
|
240
375
|
stddev: jr.caseResult.stddev,
|
|
376
|
+
// v0.6 (spec 026 AC-2, AC-8): why the generation stopped, that it
|
|
377
|
+
// was not cut (a cut draw never reaches here), and what the surface
|
|
378
|
+
// said served it (null when it said nothing).
|
|
379
|
+
stop_reason: gen.stopReason || null,
|
|
380
|
+
truncated: false,
|
|
381
|
+
reported_model: gen.attested ? modelId : null,
|
|
241
382
|
};
|
|
242
383
|
if (hasUsage(gen.usage)) draw.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
|
|
243
384
|
if (jr.caseResult.judge_usage) draw.judge_usage = jr.caseResult.judge_usage;
|
|
@@ -246,7 +387,7 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
246
387
|
if (keepTranscripts) lastTranscript = { id: c.id, mode, generation: String(gen.text || ''), judge_outputs: jr.sampleTexts };
|
|
247
388
|
} catch (e) {
|
|
248
389
|
if (e && e.code === 'BUDGET_HARDSTOP') throw e; // budget hard-stop stays fatal
|
|
249
|
-
if (!isTimeout(e)) { fatal = e; break; } // non-timeout errors stay fatal
|
|
390
|
+
if (!isTimeout(e)) { fatal = e; break; } // non-timeout errors stay fatal (a SUBSTRATE_MISMATCH among them)
|
|
250
391
|
if (budget) {
|
|
251
392
|
try {
|
|
252
393
|
if (e.phase === 'judge') budget.add((e.judgeAttempts || 1) * perCallCostUSD(judgeModel, 'judge'));
|
|
@@ -264,6 +405,9 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
264
405
|
samples: [],
|
|
265
406
|
mean: null,
|
|
266
407
|
stddev: null,
|
|
408
|
+
stop_reason: null,
|
|
409
|
+
truncated: false,
|
|
410
|
+
reported_model: null,
|
|
267
411
|
});
|
|
268
412
|
onProgress({ case: c.id, mode, phase: 'failed', draw: drawIndex, reason: String((e && e.message) || 'timeout') });
|
|
269
413
|
}
|
|
@@ -287,14 +431,22 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
287
431
|
// canary was dropped by exactly such an assembly silently gaining a field
|
|
288
432
|
// upstream that nothing here carried down (F-014-D).
|
|
289
433
|
variance_ratio_unavailable: agg.variance_ratio_unavailable,
|
|
434
|
+
// v0.6 (spec 026 AC-8): draws cut at the output cap, counted here so a
|
|
435
|
+
// reader sees it without walking the draw list.
|
|
436
|
+
n_truncated: draws.filter((d) => d.truncated === true).length,
|
|
290
437
|
draws,
|
|
291
438
|
};
|
|
292
439
|
|
|
293
|
-
// Every draw failed: the case is recorded
|
|
294
|
-
//
|
|
440
|
+
// Every draw failed: the case is recorded failed, carrying the draw list
|
|
441
|
+
// showing WHAT failed and how often. failed_timeout when every failure was
|
|
442
|
+
// a timeout; failed_unmeasured when any draw was unmeasured for another
|
|
443
|
+
// reason (spec 026 AC-3). The two share one predicate, caseFailed, in
|
|
444
|
+
// lib/receipt.js, and no reader excludes by either literal.
|
|
295
445
|
if (!last) {
|
|
296
446
|
failedCases += 1;
|
|
297
|
-
|
|
447
|
+
const status = nonTimeout ? FAILED_STATUSES[1] : FAILED_STATUSES[0];
|
|
448
|
+
const lastReason = [...draws].reverse().map((d) => d.reason).find(Boolean) || 'timeout';
|
|
449
|
+
return { caseResult: { id: c.id, mode, case_status: status, reason: lastReason, generation }, transcript: null };
|
|
298
450
|
}
|
|
299
451
|
|
|
300
452
|
// The v0.4-shaped fields now describe the DRAW SET, not one arbitrary draw,
|
|
@@ -313,7 +465,14 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
313
465
|
const caseResults = pairs.map((p) => p.caseResult);
|
|
314
466
|
const transcripts = keepTranscripts ? pairs.map((p) => p.transcript) : null;
|
|
315
467
|
|
|
316
|
-
|
|
468
|
+
// THE SURFACE IS WHAT ANSWERED, not what the runner would have chosen for
|
|
469
|
+
// the model id (spec 026 AC-1, F1). A stub run records surface stub, judge
|
|
470
|
+
// surface stub, answered_by.kind stub, and is UNVERIFIED: it measured nothing.
|
|
471
|
+
const answered = answeredByOf(replies, { attestedGen: attestedGen && replies.some((r) => r && r.answeredBy === 'model'), reportedAll });
|
|
472
|
+
const answeredBy = { kind: answered.kind, attested: answered.attested, reported_model: answered.attested ? modelId : null, reported_models: answered.reported_models, isolation: answered.isolation };
|
|
473
|
+
const surface = answered.kind === 'stub' ? 'stub' : surfaceForModel(modelId);
|
|
474
|
+
// v0.6 (spec 026 AC-11): which judge ran, and the template it graded with.
|
|
475
|
+
const judgeBlock = { ...judgeSettings(samples, judgeModel), ...(answered.kind === 'stub' ? { surface: 'stub' } : {}), model_id: judgeModel, prompt_template_hash: promptTemplateHash() };
|
|
317
476
|
const nowIso = opts.nowIso || new Date().toISOString();
|
|
318
477
|
// v0.4 economics. The pricing snapshot is frozen HERE, at run time, from the
|
|
319
478
|
// registry; every derived dollar figure below is computed from the snapshot and
|
|
@@ -355,24 +514,52 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
355
514
|
model_release_date: releaseDateFor(modelId),
|
|
356
515
|
provider: providerForModel(modelId),
|
|
357
516
|
surface,
|
|
358
|
-
// v0.3.1: on the openai/cli (codex) surface, record the fixed harness
|
|
517
|
+
// v0.3.1: on the openai/cli (codex) surface, record the fixed harness
|
|
518
|
+
// preamble. Absent on a stub run: it describes a harness that did not run.
|
|
359
519
|
surface_overhead_note: surface === 'openai-cli' ? CODEX_OVERHEAD_NOTE : undefined,
|
|
360
520
|
runner_version: RUNNER_VERSION,
|
|
361
521
|
date_utc: nowIso,
|
|
362
522
|
registry: registryStatus(modelId),
|
|
363
523
|
transcripts: keepTranscripts ? 'retained-local' : 'hashes-only',
|
|
364
|
-
judge:
|
|
524
|
+
judge: judgeBlock,
|
|
365
525
|
pricing_snapshot: pricingSnapshot,
|
|
526
|
+
answered_by: answeredBy,
|
|
366
527
|
},
|
|
367
528
|
cases: caseResults,
|
|
368
529
|
economics,
|
|
369
|
-
|
|
530
|
+
// The level is DERIVED from what answered: only a model-answered run may
|
|
531
|
+
// read TESTED. The schema refuses TESTED on a stub receipt as the second,
|
|
532
|
+
// independent control (spec 026 AC-1).
|
|
533
|
+
verificationLevel: answered.kind === 'model' ? 'TESTED' : 'UNVERIFIED',
|
|
370
534
|
});
|
|
371
535
|
|
|
372
536
|
return { receipt, calls, transcripts, failedCases };
|
|
373
537
|
}
|
|
374
538
|
|
|
375
|
-
|
|
539
|
+
// A band the formula could not form is printed as what it is, never as 0.000
|
|
540
|
+
// (spec 026 AC-7): one included case has a mean and no dispersion; none has
|
|
541
|
+
// neither.
|
|
542
|
+
function band(mean, sd) {
|
|
543
|
+
if (mean == null) return 'n/a (0 cases)';
|
|
544
|
+
if (sd == null) return `${mean.toFixed(3)} ± n/a (1 case)`;
|
|
545
|
+
return `${mean.toFixed(3)} ± ${sd.toFixed(3)}`;
|
|
546
|
+
}
|
|
547
|
+
// The comparison band, or the reason there is none.
|
|
548
|
+
function uncertaintyStr(cmp) {
|
|
549
|
+
if (cmp.delta_uncertainty != null) return cmp.delta_uncertainty.toFixed(3);
|
|
550
|
+
return `n/a (${cmp.delta_uncertainty_unavailable === 'single_case' ? '1 case' : cmp.delta_uncertainty_unavailable === 'no_cases' ? '0 cases' : 'no band'})`;
|
|
551
|
+
}
|
|
552
|
+
// The answered_by block, said in one line for the summary and the CLI.
|
|
553
|
+
function answeredLine(receipt) {
|
|
554
|
+
const ab = (receipt.run && receipt.run.answered_by) || null;
|
|
555
|
+
if (!ab) return 'answered by: unrecorded (pre-v0.6 receipt)';
|
|
556
|
+
if (ab.kind === 'stub') return 'answered by: stub (DRIFTPROOF_STUB) — nothing answered; this run measured nothing (UNVERIFIED)';
|
|
557
|
+
if (ab.kind === 'external') return 'answered by: an external tool (imported)';
|
|
558
|
+
const iso = ab.isolation === 'eval-user' ? 'the isolated eval-user hop' : ab.isolation === 'same-user' ? 'the same-user spawn (--trusted-skill)' : 'no spawn (api surface)';
|
|
559
|
+
return ab.attested
|
|
560
|
+
? `answered by: model ${ab.reported_model} (attested by the surface; ${iso})`
|
|
561
|
+
: `answered by: model, but the surface did not report which model answered (attested: false; ${iso})`;
|
|
562
|
+
}
|
|
376
563
|
|
|
377
564
|
// Render a short human-readable markdown summary of a receipt.
|
|
378
565
|
function summarizeReceipt(receipt) {
|
|
@@ -381,6 +568,7 @@ function summarizeReceipt(receipt) {
|
|
|
381
568
|
L.push('');
|
|
382
569
|
L.push(`- **model:** \`${receipt.run.model_id}\`${receipt.run.model_release_date ? ` (released ${receipt.run.model_release_date})` : ''}`);
|
|
383
570
|
L.push(`- **surface:** ${receipt.run.surface}`);
|
|
571
|
+
L.push(`- **${answeredLine(receipt)}**`);
|
|
384
572
|
L.push(`- **run (UTC):** ${receipt.run.date_utc}`);
|
|
385
573
|
L.push(`- **runner:** v${receipt.run.runner_version}`);
|
|
386
574
|
const j = receipt.run.judge || {};
|
|
@@ -395,22 +583,39 @@ function summarizeReceipt(receipt) {
|
|
|
395
583
|
L.push('');
|
|
396
584
|
const cmp = receipt.comparison;
|
|
397
585
|
const aggs = receipt.results.aggregates;
|
|
398
|
-
|
|
586
|
+
if ((receipt.run.answered_by || {}).kind === 'stub') {
|
|
587
|
+
L.push('> **STUB RUN** — DRIFTPROOF_STUB=1: nothing answered, the text was canned, and this run measured nothing. The receipt is UNVERIFIED and verdicts nothing.');
|
|
588
|
+
L.push('');
|
|
589
|
+
}
|
|
399
590
|
if (receipt.run.status === 'incomplete') {
|
|
400
591
|
L.push(`> ⏱ **INCOMPLETE** — ${receipt.run.failed_case_count} case(s) had an arm that could not be measured and are EXCLUDED from the aggregates below, BOTH arms together; this receipt must not be used to compute a drift/durability verdict.`);
|
|
401
592
|
L.push('');
|
|
402
593
|
}
|
|
594
|
+
// Spec 026 AC-8: draws cut at the output cap, said once for the run.
|
|
595
|
+
const nTruncated = receipt.results.cases.reduce((a, c) => a + (((c.generation || {}).n_truncated) || 0), 0);
|
|
596
|
+
if (nTruncated) {
|
|
597
|
+
L.push(`> ✂ **${nTruncated} draw(s) truncated** at the output cap: each is unmeasured and excluded from every band below.`);
|
|
598
|
+
L.push('');
|
|
599
|
+
}
|
|
403
600
|
L.push(`with_skill **${band(aggs.with_skill.mean_score, aggs.with_skill.stddev)}** vs baseline **${band(aggs.baseline.mean_score, aggs.baseline.stddev)}**`);
|
|
404
601
|
L.push('');
|
|
405
|
-
|
|
602
|
+
if (cmp.delta == null) {
|
|
603
|
+
L.push(`skill lift **n/a** (${cmp.delta_uncertainty_unavailable === 'no_cases' ? 'no case was included on an arm' : 'no comparison'})`);
|
|
604
|
+
} else {
|
|
605
|
+
const sign = cmp.delta >= 0 ? '+' : '';
|
|
606
|
+
L.push(`skill lift **${sign}${cmp.delta.toFixed(3)}** (combined uncertainty ± ${uncertaintyStr(cmp)})`);
|
|
607
|
+
}
|
|
608
|
+
L.push('');
|
|
609
|
+
// The rule the bands above are derived by, said beside them (spec 026 AC-6).
|
|
610
|
+
L.push(`band rule: each arm's band is the sample stddev of its per-case means${aggs.band_rule ? ` — ${aggs.band_rule}` : ' (unstated on this pre-v0.6 receipt)'}`);
|
|
406
611
|
L.push('');
|
|
407
612
|
L.push(`## Per-case (mean ± stddev over ${(receipt.run.judge || {}).samples || 1} judge samples)`);
|
|
408
613
|
L.push('');
|
|
409
614
|
L.push(`| case | mode | outcome | mean ± stddev | judge reason |`);
|
|
410
615
|
L.push(`|---|---|---|---|---|`);
|
|
411
616
|
for (const c of receipt.results.cases) {
|
|
412
|
-
if (c
|
|
413
|
-
L.push(`| \`${c.id}\` | ${c.mode} | ⏱
|
|
617
|
+
if (caseFailed(c)) {
|
|
618
|
+
L.push(`| \`${c.id}\` | ${c.mode} | ⏱ ${c.case_status} | — (not measured) | ${c.reason || 'not measured'} |`);
|
|
414
619
|
continue;
|
|
415
620
|
}
|
|
416
621
|
const flag = c.outcome === 'borderline' ? ' ⚠' : '';
|
|
@@ -420,4 +625,4 @@ function summarizeReceipt(receipt) {
|
|
|
420
625
|
return L.join('\n');
|
|
421
626
|
}
|
|
422
627
|
|
|
423
|
-
module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor, resolveCallTimeoutMs };
|
|
628
|
+
module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor, resolveCallTimeoutMs, answeredLine, answeredByOf, attest, band, uncertaintyStr };
|
package/lib/skill.js
CHANGED
|
@@ -4,7 +4,16 @@
|
|
|
4
4
|
const fs = require('fs');
|
|
5
5
|
const path = require('path');
|
|
6
6
|
const { sha256Files, sha256Canonical } = require('./canonical');
|
|
7
|
-
const { SUITE_FORMAT } = require('../config');
|
|
7
|
+
const { SUITE_FORMAT, SKILL_MAX_FILES, SKILL_MAX_BYTES, SKILL_MAX_DEPTH, SUITE_MAX_CASES, CASE_MAX_CHARS } = require('../config');
|
|
8
|
+
|
|
9
|
+
// A bound was exceeded: the error names the bound and the value (spec 026
|
|
10
|
+
// AC-13), so the refusal says what to change rather than that something is
|
|
11
|
+
// too big.
|
|
12
|
+
function pastBound(bound, limit, value, what) {
|
|
13
|
+
const e = new Error(`${what}: ${value} exceeds ${bound} (${limit}); refusing to load`);
|
|
14
|
+
e.code = 'INPUT_BOUND'; e.bound = bound; e.limit = limit; e.value = value;
|
|
15
|
+
return e;
|
|
16
|
+
}
|
|
8
17
|
|
|
9
18
|
// Load a skill directory and its eval suite.
|
|
10
19
|
//
|
|
@@ -21,13 +30,22 @@ const { SUITE_FORMAT } = require('../config');
|
|
|
21
30
|
const IGNORE_DIRS = new Set(['.git', 'node_modules', 'evals']);
|
|
22
31
|
const IGNORE_FILES = new Set(['.DS_Store']);
|
|
23
32
|
|
|
24
|
-
|
|
33
|
+
// Bounded (spec 026 AC-13): the walk stops at SKILL_MAX_DEPTH directories
|
|
34
|
+
// below the skill dir, SKILL_MAX_FILES bundled files, and SKILL_MAX_BYTES of
|
|
35
|
+
// them together, and throws naming the bound the moment one is passed, so a
|
|
36
|
+
// pathological tree is refused before its bytes are read into memory.
|
|
37
|
+
function walkFiles(dir, base = dir, acc = [], state = { bytes: 0 }, depth = 0) {
|
|
38
|
+
if (depth > SKILL_MAX_DEPTH) throw pastBound('SKILL_MAX_DEPTH', SKILL_MAX_DEPTH, depth, `directory depth under ${base}`);
|
|
25
39
|
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
|
26
40
|
if (entry.isDirectory()) {
|
|
27
41
|
if (IGNORE_DIRS.has(entry.name)) continue;
|
|
28
|
-
walkFiles(path.join(dir, entry.name), base, acc);
|
|
42
|
+
walkFiles(path.join(dir, entry.name), base, acc, state, depth + 1);
|
|
29
43
|
} else if (entry.isFile() && !IGNORE_FILES.has(entry.name)) {
|
|
30
44
|
const abs = path.join(dir, entry.name);
|
|
45
|
+
if (acc.length + 1 > SKILL_MAX_FILES) throw pastBound('SKILL_MAX_FILES', SKILL_MAX_FILES, acc.length + 1, `bundled files under ${base}`);
|
|
46
|
+
const size = fs.statSync(abs).size;
|
|
47
|
+
state.bytes += size;
|
|
48
|
+
if (state.bytes > SKILL_MAX_BYTES) throw pastBound('SKILL_MAX_BYTES', SKILL_MAX_BYTES, state.bytes, `bundled bytes under ${base}`);
|
|
31
49
|
acc.push({ path: path.relative(base, abs), bytes: fs.readFileSync(abs) });
|
|
32
50
|
}
|
|
33
51
|
}
|
|
@@ -108,12 +126,16 @@ function normalizeCases(raw) {
|
|
|
108
126
|
: Array.isArray(raw.evals) ? raw.evals
|
|
109
127
|
: null;
|
|
110
128
|
if (!list) throw new Error('evals.json must be an array or have a `cases`/`evals` array');
|
|
129
|
+
// Bounded (spec 026 AC-13): the case count, and each prompt and rubric.
|
|
130
|
+
if (list.length > SUITE_MAX_CASES) throw pastBound('SUITE_MAX_CASES', SUITE_MAX_CASES, list.length, 'cases in the suite');
|
|
111
131
|
return list.map((c, i) => {
|
|
112
132
|
const id = String(c.id || c.name || `case-${i + 1}`);
|
|
113
133
|
const prompt = c.prompt || c.input || c.task;
|
|
114
134
|
const rubric = c.rubric || c.criteria || c.expected;
|
|
115
135
|
if (!prompt) throw new Error(`case "${id}" is missing a prompt/input/task`);
|
|
116
136
|
if (!rubric) throw new Error(`case "${id}" is missing a rubric/criteria/expected`);
|
|
137
|
+
if (String(prompt).length > CASE_MAX_CHARS) throw pastBound('CASE_MAX_CHARS', CASE_MAX_CHARS, String(prompt).length, `case "${id}" prompt characters`);
|
|
138
|
+
if (String(rubric).length > CASE_MAX_CHARS) throw pastBound('CASE_MAX_CHARS', CASE_MAX_CHARS, String(rubric).length, `case "${id}" rubric characters`);
|
|
117
139
|
const threshold = typeof c.pass_threshold === 'number' ? c.pass_threshold
|
|
118
140
|
: typeof c.threshold === 'number' ? c.threshold : 0.7;
|
|
119
141
|
const norm = { id, prompt: String(prompt), rubric: String(rubric), pass_threshold: threshold };
|
|
@@ -124,4 +146,4 @@ function normalizeCases(raw) {
|
|
|
124
146
|
});
|
|
125
147
|
}
|
|
126
148
|
|
|
127
|
-
module.exports = { loadSkill, normalizeCases, parseSkillMeta };
|
|
149
|
+
module.exports = { loadSkill, normalizeCases, parseSkillMeta, pastBound };
|
package/lib/stats.js
CHANGED
|
@@ -31,8 +31,11 @@ function stderr(xs) {
|
|
|
31
31
|
return round(stddev(xs) / Math.sqrt(n));
|
|
32
32
|
}
|
|
33
33
|
|
|
34
|
-
// Combine independent uncertainties in quadrature: sqrt(a^2 + b^2).
|
|
34
|
+
// Combine independent uncertainties in quadrature: sqrt(a^2 + b^2). Null when
|
|
35
|
+
// either band is null: a combination of a band that could not form cannot form
|
|
36
|
+
// either (spec 026 AC-7), and the receipt says why beside it.
|
|
35
37
|
function combineUncertainty(a, b) {
|
|
38
|
+
if (a == null || b == null) return null;
|
|
36
39
|
return round(Math.sqrt(a * a + b * b));
|
|
37
40
|
}
|
|
38
41
|
|
|
@@ -45,10 +48,16 @@ function combineUncertainty(a, b) {
|
|
|
45
48
|
// is a conventional, honest "mean ± stddev across the suite". The drift HEADLINE
|
|
46
49
|
// verdict is driven by the per-case band-overlap verdicts (see lib/diff.js), not
|
|
47
50
|
// by this aggregate band; this value is a reported summary statistic.
|
|
51
|
+
//
|
|
52
|
+
// A BAND THE FORMULA CANNOT FORM IS NULL, NEVER 0 (spec 026 AC-7, F3). The
|
|
53
|
+
// sample standard deviation of one value is undefined, and the mean of no
|
|
54
|
+
// values is not a number; printing 0.000 for either asserted a precision that
|
|
55
|
+
// was never measured (the one-case run's `± 0.000`). The rule the receipt
|
|
56
|
+
// states (results.aggregates.band_rule) is this function.
|
|
48
57
|
function aggregateBands(cases) {
|
|
49
|
-
if (!cases.length) return { mean:
|
|
58
|
+
if (!cases.length) return { mean: null, stddev: null };
|
|
50
59
|
const means = cases.map((c) => c.mean);
|
|
51
|
-
return { mean: mean(means), stddev: stddev(means) };
|
|
60
|
+
return { mean: mean(means), stddev: means.length < 2 ? null : stddev(means) };
|
|
52
61
|
}
|
|
53
62
|
|
|
54
63
|
// Do two confidence bands (mean ± half-width) fail to overlap, and in which
|
package/lib/usage.js
CHANGED
|
@@ -42,6 +42,27 @@
|
|
|
42
42
|
|
|
43
43
|
function n(x) { return Number.isFinite(Number(x)) ? Number(x) : 0; }
|
|
44
44
|
|
|
45
|
+
// ── what answered (spec 026, AC-2 and AC-8) ──────────────────────────────────
|
|
46
|
+
// Every lane also reports, when it can, WHICH model served the call and WHY the
|
|
47
|
+
// reply stopped. Both are read here, never guessed: a surface that says nothing
|
|
48
|
+
// yields null for both, and the receipt records that it said nothing.
|
|
49
|
+
//
|
|
50
|
+
// reportedModels the ids the surface named, VERBATIM (the real claude CLI
|
|
51
|
+
// keys the call's usage under the undated form and puts a
|
|
52
|
+
// dated side entry beside it; an api response names one
|
|
53
|
+
// model). Null when the surface names no model. The runner
|
|
54
|
+
// compares on canonical ids (canonicalModelId: a trailing
|
|
55
|
+
// -YYYYMMDD is not part of the identity), never the reader.
|
|
56
|
+
// stopReason the surface's own word for why the reply ended (end_turn,
|
|
57
|
+
// max_tokens, stop, length ...). Null when it gives none.
|
|
58
|
+
function canonicalModelId(id) { return String(id || '').replace(/-\d{8}$/, ''); }
|
|
59
|
+
function reportedModelsOf(map) {
|
|
60
|
+
if (!map || typeof map !== 'object' || Array.isArray(map)) return null;
|
|
61
|
+
const ids = [...new Set(Object.keys(map).filter((k) => k))].sort();
|
|
62
|
+
return ids.length ? ids : null;
|
|
63
|
+
}
|
|
64
|
+
function stopReasonOf(v) { return typeof v === 'string' && v ? v : null; }
|
|
65
|
+
|
|
45
66
|
// The empty/unknown usage record. Deliberately null (not zeros) so "the surface
|
|
46
67
|
// did not tell us" never reads as "the call cost nothing".
|
|
47
68
|
function emptyUsage() {
|
|
@@ -75,6 +96,9 @@ function parseClaudeCliJson(stdout) {
|
|
|
75
96
|
text: typeof j.result === 'string' ? j.result : '',
|
|
76
97
|
usage,
|
|
77
98
|
isError: j.is_error === true,
|
|
99
|
+
// v0.6: the CLI's own stop reason and the models it says served the call.
|
|
100
|
+
stopReason: stopReasonOf(j.stop_reason),
|
|
101
|
+
reportedModels: reportedModelsOf(j.modelUsage),
|
|
78
102
|
// The CLI reports its own dollar figure. Recorded here for completeness but
|
|
79
103
|
// NOT used: costs are computed uniformly from the frozen pricing snapshot so
|
|
80
104
|
// three substrates are on one basis (see lib/value.js).
|
|
@@ -85,7 +109,9 @@ function parseClaudeCliJson(stdout) {
|
|
|
85
109
|
// ── openai/cli — `codex exec --json` JSONL event stream ───────────────────────
|
|
86
110
|
// One JSON object per line. Usage rides the terminal `turn.completed` event;
|
|
87
111
|
// input_tokens already INCLUDES cached_input_tokens. Unparseable lines are
|
|
88
|
-
// skipped (the stream also carries progress events we do not model).
|
|
112
|
+
// skipped (the stream also carries progress events we do not model). The
|
|
113
|
+
// stream names neither the model that served the turn nor a stop reason, so
|
|
114
|
+
// both read null here and the receipt says the surface did not report them.
|
|
89
115
|
function parseCodexJsonl(stdout) {
|
|
90
116
|
const lines = String(stdout || '').split('\n').map((l) => l.trim()).filter(Boolean);
|
|
91
117
|
let usage = null;
|
|
@@ -103,16 +129,30 @@ function parseCodexJsonl(stdout) {
|
|
|
103
129
|
wall_ms: null,
|
|
104
130
|
};
|
|
105
131
|
}
|
|
106
|
-
// The final message
|
|
132
|
+
// The final message IS the stream: the last `item.completed` /
|
|
133
|
+
// `agent_message` event carries it. Nothing is read from a file after the
|
|
134
|
+
// call, on either spawn mode (spec 022 AC-9; the earlier comment here
|
|
135
|
+
// described an output file the lane no longer reads, F-022-3).
|
|
107
136
|
if (ev.type === 'item.completed' && ev.item && ev.item.type === 'agent_message' && typeof ev.item.text === 'string') {
|
|
108
137
|
text = ev.item.text;
|
|
109
138
|
}
|
|
110
139
|
}
|
|
111
|
-
return { usage, text };
|
|
140
|
+
return { usage, text, stopReason: null, reportedModels: null };
|
|
112
141
|
}
|
|
113
142
|
|
|
114
143
|
// ── anthropic/api ─────────────────────────────────────────────────────────────
|
|
115
144
|
// The Messages API reports input_tokens EXCLUDING cache, same as the CLI.
|
|
145
|
+
// The response object also names the model that served it (`model`) and why
|
|
146
|
+
// it stopped (`stop_reason`); read here from the object, null when absent.
|
|
147
|
+
// Exercised against tests/fixtures/response-anthropic-api.json, which is
|
|
148
|
+
// SYNTHESISED from the API reference, not captured from a call.
|
|
149
|
+
function readAnthropicApiResponse(resp) {
|
|
150
|
+
const r = resp && typeof resp === 'object' ? resp : {};
|
|
151
|
+
return {
|
|
152
|
+
stopReason: stopReasonOf(r.stop_reason),
|
|
153
|
+
reportedModels: typeof r.model === 'string' && r.model ? [r.model] : null,
|
|
154
|
+
};
|
|
155
|
+
}
|
|
116
156
|
function parseAnthropicApiUsage(u) {
|
|
117
157
|
if (!u) return emptyUsage();
|
|
118
158
|
const cacheRead = n(u.cache_read_input_tokens);
|
|
@@ -127,7 +167,18 @@ function parseAnthropicApiUsage(u) {
|
|
|
127
167
|
|
|
128
168
|
// ── openai/api (Chat Completions-compatible) ──────────────────────────────────
|
|
129
169
|
// prompt_tokens is the total; the cached portion, when present, is nested under
|
|
130
|
-
// prompt_tokens_details.cached_tokens.
|
|
170
|
+
// prompt_tokens_details.cached_tokens. The response also names the serving
|
|
171
|
+
// model (`model`) and the first choice's `finish_reason`; read here, null
|
|
172
|
+
// when absent. Exercised against tests/fixtures/response-openai-api.json,
|
|
173
|
+
// SYNTHESISED from the API reference, not captured from a call.
|
|
174
|
+
function readOpenaiApiResponse(resp) {
|
|
175
|
+
const r = resp && typeof resp === 'object' ? resp : {};
|
|
176
|
+
const choice = Array.isArray(r.choices) && r.choices[0] && typeof r.choices[0] === 'object' ? r.choices[0] : {};
|
|
177
|
+
return {
|
|
178
|
+
stopReason: stopReasonOf(choice.finish_reason),
|
|
179
|
+
reportedModels: typeof r.model === 'string' && r.model ? [r.model] : null,
|
|
180
|
+
};
|
|
181
|
+
}
|
|
131
182
|
function parseOpenaiApiUsage(u) {
|
|
132
183
|
if (!u) return emptyUsage();
|
|
133
184
|
const details = u.prompt_tokens_details || {};
|
|
@@ -165,4 +216,5 @@ function normalizeUsage(u) {
|
|
|
165
216
|
module.exports = {
|
|
166
217
|
emptyUsage, hasUsage, sumUsage, normalizeUsage,
|
|
167
218
|
parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage,
|
|
219
|
+
readAnthropicApiResponse, readOpenaiApiResponse, canonicalModelId, reportedModelsOf,
|
|
168
220
|
};
|