driftproof 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/run.js CHANGED
@@ -1,12 +1,16 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
- const { complete, resolveModel, surfaceLabel } = require('./provider');
4
+ const { complete, resolveModel, surfaceForModel, isMeteredSurface, CODEX_OVERHEAD_NOTE } = require('./provider');
5
5
  const { gradeSamples, judgeSettings } = require('./judge');
6
6
  const { buildReceipt } = require('./receipt');
7
7
  const { sha256 } = require('./canonical');
8
- const { registryStatus } = require('./models');
8
+ const { registryStatus, providerForModel, priceForModel } = require('./models');
9
9
  const { perCallCostUSD } = require('./cost');
10
+ const { runChecks } = require('./checks');
11
+ const { estimateTokens } = require('./skillCost');
12
+ const { hasUsage, normalizeUsage } = require('./usage');
13
+ const { buildPricingSnapshot, computeEconomics } = require('./value');
10
14
  const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES } = require('../config');
11
15
 
12
16
  // Known model release dates (best-effort; null when unknown). Recorded into the
@@ -20,6 +24,8 @@ const MODEL_RELEASE_DATES = {
20
24
  'claude-haiku-4-5': '2025-10-01',
21
25
  'claude-sonnet-5': '2026-06-30', // anthropic.com/news/claude-sonnet-5
22
26
  'claude-sonnet-4-6': '2026-02-17', // anthropic.com/news/claude-sonnet-4-6
27
+ 'claude-opus-5': '2026-07-24', // anthropic.com/news/claude-opus-5
28
+ 'claude-opus-4-8': '2026-05-28', // anthropic.com/news/claude-opus-4-8
23
29
  };
24
30
 
25
31
  function releaseDateFor(modelId) {
@@ -39,9 +45,16 @@ function projectCalls(caseCount, samples) {
39
45
  // SKILL.md is prepended as a system prompt (the whole point: measure the skill's
40
46
  // marginal effect vs a bare baseline).
41
47
  async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs }) {
48
+ // Test seam (gate only): force a persistent timeout for a named case id so the
49
+ // failed_timeout path is exercised deterministically without any live call.
50
+ if (process.env.DRIFTPROOF_TEST_TIMEOUT_CASEID && process.env.DRIFTPROOF_TEST_TIMEOUT_CASEID === caseObj.id) {
51
+ const e = new Error('provider timed out (test seam) after retries');
52
+ e.code = 'TIMEOUT'; e.attempts = 4;
53
+ throw e;
54
+ }
42
55
  const system = withSkill ? skillMd : undefined;
43
- const { text, usage } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
44
- return { text, usage };
56
+ const { text, usage, wall_ms, attempts } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
57
+ return { text, usage, wall_ms, attempts };
45
58
  }
46
59
 
47
60
  // Determine a case outcome from its sampled band and threshold.
@@ -77,7 +90,12 @@ async function judgeCase({ caseObj, response, generationHash, judgeModel, mode,
77
90
  reason: g.reason,
78
91
  judge: { model_id: g.model_id, rubric_hash: g.rubric_hash },
79
92
  };
80
- return { caseResult, sampleTexts: g.sample_texts };
93
+ // v0.3.1 deterministic post-checks (supplementary; NOT folded into `outcome`).
94
+ const checks = runChecks(response, caseObj.checks);
95
+ if (checks.length) caseResult.checks = checks;
96
+ // v0.4: grading overhead for this case row, kept OUT of the skill-value math.
97
+ if (hasUsage(g.usage)) caseResult.judge_usage = g.usage;
98
+ return { caseResult, sampleTexts: g.sample_texts, attempts: g.attempts };
81
99
  }
82
100
 
83
101
  // Run up to `concurrency` async tasks at a time, preserving input order in the
@@ -112,6 +130,9 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
112
130
  const modelId = resolveModel(model);
113
131
  const judgeModel = opts.judgeModel ? resolveModel(opts.judgeModel) : modelId;
114
132
  const timeoutMs = opts.timeoutMs || 120000;
133
+ // Optional per-case timeout overrides { caseId: ms }; a slow case can get a
134
+ // longer budget without lengthening every other case's per-call timeout.
135
+ const caseTimeoutMs = opts.caseTimeoutMs || {};
115
136
  const maxCalls = opts.maxCalls || 200;
116
137
  const samples = opts.samples || DEFAULT_JUDGE_SAMPLES;
117
138
  const concurrency = Math.max(1, opts.concurrency || 1);
@@ -137,46 +158,98 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
137
158
  for (const c of cases) for (const withSkill of [true, false]) tasks.push({ c, withSkill });
138
159
 
139
160
  let calls = 0;
161
+ let failedCases = 0;
162
+ const isTimeout = (e) => !!(e && (e.code === 'TIMEOUT' || /tim(e|ed)\s*out/i.test(String((e && e.message) || ''))));
163
+ const genKind = (ws) => (ws ? 'gen_with_skill' : 'gen_baseline');
140
164
  const pairs = await mapPool(tasks, concurrency, async ({ c, withSkill }) => {
141
165
  const mode = withSkill ? 'with_skill' : 'baseline';
142
- onProgress({ case: c.id, mode, phase: 'generate' });
143
- const { text } = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs });
144
- calls += 1;
145
- // Live budget: count the generation call, then hard-stop if over 1.25× cap.
146
- if (budget) budget.add(perCallCostUSD(modelId, withSkill ? 'gen_with_skill' : 'gen_baseline'));
147
- const generationHash = sha256(String(text || ''));
148
- onProgress({ case: c.id, mode, phase: 'judge', samples });
149
- const { caseResult, sampleTexts } = await judgeCase({ caseObj: c, response: text, generationHash, judgeModel, mode, timeoutMs, samples });
150
- calls += samples;
151
- // Live budget: count all `samples` judge calls for this (case, mode).
152
- if (budget) budget.add(samples * perCallCostUSD(judgeModel, 'judge'));
153
- onProgress({ case: c.id, mode, phase: 'done', outcome: caseResult.outcome, score: caseResult.mean, stddev: caseResult.stddev });
154
- const transcript = keepTranscripts
155
- ? { id: c.id, mode, generation: String(text || ''), judge_outputs: sampleTexts }
156
- : null;
157
- return { caseResult, transcript };
166
+ const ct = caseTimeoutMs[c.id] || timeoutMs;
167
+ try {
168
+ onProgress({ case: c.id, mode, phase: 'generate' });
169
+ const gen = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs: ct });
170
+ calls += 1;
171
+ // Live budget: count the generation call INCLUDING retries, then hard-stop
172
+ // if over 1.25× cap.
173
+ if (budget) budget.add((gen.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
174
+ const generationHash = sha256(String(gen.text || ''));
175
+ onProgress({ case: c.id, mode, phase: 'judge', samples });
176
+ const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ct, samples });
177
+ // v0.4: the GENERATION call's usage is the skill-value measurement (the
178
+ // judge's own usage rides separately on judge_usage, above).
179
+ if (hasUsage(gen.usage)) jr.caseResult.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
180
+ calls += samples;
181
+ // Live budget: count all judge calls (retries included) for this (case, mode).
182
+ if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judgeModel, 'judge'));
183
+ onProgress({ case: c.id, mode, phase: 'done', outcome: jr.caseResult.outcome, score: jr.caseResult.mean, stddev: jr.caseResult.stddev });
184
+ const transcript = keepTranscripts
185
+ ? { id: c.id, mode, generation: String(gen.text || ''), judge_outputs: jr.sampleTexts }
186
+ : null;
187
+ return { caseResult: jr.caseResult, transcript };
188
+ } catch (e) {
189
+ if (e && e.code === 'BUDGET_HARDSTOP') throw e; // budget hard-stop stays fatal
190
+ if (!isTimeout(e)) throw e; // non-timeout errors stay fatal
191
+ // Persistent timeout → NON-FATAL: charge the consumed attempts, record the
192
+ // case as failed_timeout (no fabricated samples), and continue the run.
193
+ if (budget) {
194
+ try {
195
+ if (e.phase === 'judge') budget.add((e.judgeAttempts || 1) * perCallCostUSD(judgeModel, 'judge'));
196
+ else budget.add((e.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
197
+ } catch (be) { if (be && be.code === 'BUDGET_HARDSTOP') throw be; }
198
+ }
199
+ failedCases += 1;
200
+ onProgress({ case: c.id, mode, phase: 'failed', reason: String((e && e.message) || 'timeout') });
201
+ return { caseResult: { id: c.id, mode, case_status: 'failed_timeout', reason: String((e && e.message) || 'timeout').slice(0, 200) }, transcript: null };
202
+ }
158
203
  });
159
204
  const caseResults = pairs.map((p) => p.caseResult);
160
205
  const transcripts = keepTranscripts ? pairs.map((p) => p.transcript) : null;
161
206
 
207
+ const surface = surfaceForModel(modelId);
208
+ const nowIso = opts.nowIso || new Date().toISOString();
209
+ // v0.4 economics. The pricing snapshot is frozen HERE, at run time, from the
210
+ // registry; every derived dollar figure below is computed from the snapshot and
211
+ // never from the live registry, so this receipt keeps its meaning when prices
212
+ // later change.
213
+ const pricingSnapshot = buildPricingSnapshot({
214
+ models: [modelId, judgeModel],
215
+ lookup: priceForModel,
216
+ nowIso,
217
+ });
218
+ const economics = computeEconomics({
219
+ cases: caseResults,
220
+ modelId,
221
+ judgeModelId: judgeModel,
222
+ pricingSnapshot,
223
+ surface,
224
+ meteredSurface: isMeteredSurface(surface),
225
+ });
162
226
  const receipt = buildReceipt({
163
- skill: { name: skill.name, version: skill.version, contentHash: skill.contentHash },
227
+ skill: {
228
+ name: skill.name, version: skill.version, contentHash: skill.contentHash,
229
+ // v0.3.1 value-per-token axis: estimated SKILL.md token size.
230
+ tokens: estimateTokens(skill.skillMd),
231
+ },
164
232
  suite: { format: skill.suite.format, suiteHash: skill.suite.suiteHash, caseCount: skill.suite.caseCount },
165
233
  run: {
166
234
  model_id: modelId,
167
235
  model_release_date: releaseDateFor(modelId),
168
- surface: surfaceLabel(),
236
+ provider: providerForModel(modelId),
237
+ surface,
238
+ // v0.3.1: on the openai/cli (codex) surface, record the fixed harness preamble.
239
+ surface_overhead_note: surface === 'openai-cli' ? CODEX_OVERHEAD_NOTE : undefined,
169
240
  runner_version: RUNNER_VERSION,
170
- date_utc: opts.nowIso || new Date().toISOString(),
241
+ date_utc: nowIso,
171
242
  registry: registryStatus(modelId),
172
243
  transcripts: keepTranscripts ? 'retained-local' : 'hashes-only',
173
- judge: judgeSettings(samples),
244
+ judge: judgeSettings(samples, judgeModel),
245
+ pricing_snapshot: pricingSnapshot,
174
246
  },
175
247
  cases: caseResults,
248
+ economics,
176
249
  verificationLevel: 'TESTED',
177
250
  });
178
251
 
179
- return { receipt, calls, transcripts };
252
+ return { receipt, calls, transcripts, failedCases };
180
253
  }
181
254
 
182
255
  function band(mean, sd) { return `${mean.toFixed(3)} ± ${sd.toFixed(3)}`; }
@@ -191,7 +264,7 @@ function summarizeReceipt(receipt) {
191
264
  L.push(`- **run (UTC):** ${receipt.run.date_utc}`);
192
265
  L.push(`- **runner:** v${receipt.run.runner_version}`);
193
266
  const j = receipt.run.judge || {};
194
- L.push(`- **judge:** ${j.samples || 1} samples/case, temperature ${j.temperature == null ? 'n/a (surface-controlled)' : j.temperature} (${j.sampling || 'single'})`);
267
+ L.push(`- **judge:** ${j.samples || 1} samples/case, temperature ${j.temperature == null ? 'n/a' : j.temperature} (${j.sampling || 'single'})`);
195
268
  if (receipt.run.registry) L.push(`- **registry:** ${receipt.run.registry} **transcripts:** ${receipt.run.transcripts || 'hashes-only'}`);
196
269
  L.push(`- **skill content_hash:** \`${receipt.skill.content_hash.slice(0, 16)}…\``);
197
270
  L.push(`- **suite:** ${receipt.suite.case_count} cases (${receipt.suite.format})`);
@@ -203,6 +276,10 @@ function summarizeReceipt(receipt) {
203
276
  const cmp = receipt.comparison;
204
277
  const aggs = receipt.results.aggregates;
205
278
  const sign = cmp.delta >= 0 ? '+' : '';
279
+ if (receipt.run.status === 'incomplete') {
280
+ L.push(`> ⏱ **INCOMPLETE** — ${receipt.run.failed_case_count} case(s) failed (timed out after retries) and are EXCLUDED from the aggregates below; this receipt must not be used to compute a drift/durability verdict.`);
281
+ L.push('');
282
+ }
206
283
  L.push(`with_skill **${band(aggs.with_skill.mean_score, aggs.with_skill.stddev)}** vs baseline **${band(aggs.baseline.mean_score, aggs.baseline.stddev)}**`);
207
284
  L.push('');
208
285
  L.push(`skill lift **${sign}${cmp.delta.toFixed(3)}** (combined uncertainty ± ${cmp.delta_uncertainty.toFixed(3)})`);
@@ -212,6 +289,10 @@ function summarizeReceipt(receipt) {
212
289
  L.push(`| case | mode | outcome | mean ± stddev | judge reason |`);
213
290
  L.push(`|---|---|---|---|---|`);
214
291
  for (const c of receipt.results.cases) {
292
+ if (c.case_status === 'failed_timeout') {
293
+ L.push(`| \`${c.id}\` | ${c.mode} | ⏱ failed_timeout | — (not measured) | ${c.reason || 'timed out'} |`);
294
+ continue;
295
+ }
215
296
  const flag = c.outcome === 'borderline' ? ' ⚠' : '';
216
297
  L.push(`| \`${c.id}\` | ${c.mode} | ${c.outcome}${flag} | ${band(c.mean, c.stddev || 0)} | ${c.reason || ''} |`);
217
298
  }
package/lib/skill.js CHANGED
@@ -59,7 +59,13 @@ function loadSkill(skillDir) {
59
59
  }
60
60
  const suiteRaw = JSON.parse(fs.readFileSync(suitePath, 'utf8'));
61
61
  const cases = normalizeCases(suiteRaw);
62
- const suiteHash = sha256Canonical(cases);
62
+ // suite_hash is over the CORE case fields only ({id, prompt, rubric,
63
+ // pass_threshold}); optional annotations (checks, and the claim/grounding the
64
+ // gate reads straight from the raw suite) are excluded so adding them never
65
+ // disturbs a receipt's suite_hash. See spec/RECEIPT.md § suite.
66
+ const suiteHash = sha256Canonical(cases.map((c) => ({
67
+ id: c.id, prompt: c.prompt, rubric: c.rubric, pass_threshold: c.pass_threshold,
68
+ })));
63
69
 
64
70
  return {
65
71
  dir,
@@ -110,7 +116,11 @@ function normalizeCases(raw) {
110
116
  if (!rubric) throw new Error(`case "${id}" is missing a rubric/criteria/expected`);
111
117
  const threshold = typeof c.pass_threshold === 'number' ? c.pass_threshold
112
118
  : typeof c.threshold === 'number' ? c.threshold : 0.7;
113
- return { id, prompt: String(prompt), rubric: String(rubric), pass_threshold: threshold };
119
+ const norm = { id, prompt: String(prompt), rubric: String(rubric), pass_threshold: threshold };
120
+ // Optional deterministic post-checks (v0.3.1). Carried through for the runner
121
+ // but EXCLUDED from suite_hash (see loadSkill) so they never disturb a receipt.
122
+ if (Array.isArray(c.checks) && c.checks.length) norm.checks = c.checks;
123
+ return norm;
114
124
  });
115
125
  }
116
126
 
@@ -0,0 +1,31 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // Value-per-token axis.
5
+ //
6
+ // A skill is not free: its SKILL.md is prepended to every generation, so a skill
7
+ // that lifts scores by +0.10 for 400 tokens is a very different proposition from
8
+ // one that lifts +0.10 for 6,000 tokens. Driftproof reports each skill's lift
9
+ // (delta) BOTH raw AND normalized per 1,000 skill tokens.
10
+ //
11
+ // Token estimate: a coarse `ceil(chars / 4)` proxy — the standard rough rule for
12
+ // English text, NOT a model tokenizer (no tiktoken dependency; the number is a
13
+ // consistent proxy for comparing skills, not a billing figure). Documented on
14
+ // docs/methodology.html.
15
+
16
+ const CHARS_PER_TOKEN = 4;
17
+
18
+ function estimateTokens(text) {
19
+ const s = String(text || '');
20
+ if (!s.length) return 0;
21
+ return Math.ceil(s.length / CHARS_PER_TOKEN);
22
+ }
23
+
24
+ // Lift per 1,000 skill tokens: delta / (skillTokens / 1000). Null when the token
25
+ // count is unknown or zero (avoids a divide-by-zero masquerading as infinite value).
26
+ function deltaPer1kTokens(delta, skillTokens) {
27
+ if (typeof delta !== 'number' || !skillTokens || skillTokens <= 0) return null;
28
+ return Math.round((delta / (skillTokens / 1000)) * 1e4) / 1e4;
29
+ }
30
+
31
+ module.exports = { estimateTokens, deltaPer1kTokens, CHARS_PER_TOKEN };
package/lib/stub.js CHANGED
@@ -39,16 +39,37 @@ function stubComplete({ system, prompt }) {
39
39
  pass: score >= 0.7,
40
40
  reason: `stub judge (DRIFTPROOF_STUB): ${helped ? 'with-skill marker present' : 'baseline'}`,
41
41
  };
42
- return { text: JSON.stringify(body), usage: null };
42
+ return { text: JSON.stringify(body), usage: stubUsage({ kind: 'judge', prompt }) };
43
43
  }
44
44
  // Generation call: canned, marked by mode.
45
45
  const mark = system ? MARK_SKILL : MARK_BASE;
46
46
  const text = `${mark}\nfeat(stub): canned offline generation for CI (no model was called)`;
47
- return { text, usage: null };
47
+ return { text, usage: stubUsage({ kind: 'gen', system, prompt }) };
48
+ }
49
+
50
+ // v0.4: deterministic synthetic usage, so a zero-model-call stub run still
51
+ // exercises the whole economics path (usage → pricing snapshot → derived cost/
52
+ // latency fields → the value report's three axes). The numbers are SYNTHETIC and
53
+ // deliberately shaped like the real surfaces: a large fixed harness preamble that
54
+ // is identical in both arms, plus the with-skill arm's extra SKILL.md input and
55
+ // its slightly longer, slower output. No randomness — same input, same usage.
56
+ const STUB_HARNESS_PREAMBLE_TOKENS = 20000; // the fixed, arm-identical overhead
57
+ function stubUsage({ kind, system, prompt }) {
58
+ const promptTokens = Math.ceil(String(prompt || '').length / 4);
59
+ if (kind === 'judge') {
60
+ return { input_tokens: 1200 + promptTokens, output_tokens: 60, cached_tokens: 800, wall_ms: 900 };
61
+ }
62
+ const skillTokens = system ? Math.ceil(String(system).length / 4) : 0;
63
+ return {
64
+ input_tokens: STUB_HARNESS_PREAMBLE_TOKENS + promptTokens + skillTokens,
65
+ output_tokens: system ? 420 : 350,
66
+ cached_tokens: STUB_HARNESS_PREAMBLE_TOKENS,
67
+ wall_ms: system ? 4200 : 3600,
68
+ };
48
69
  }
49
70
 
50
71
  function stubEnabled() {
51
72
  return process.env.DRIFTPROOF_STUB === '1';
52
73
  }
53
74
 
54
- module.exports = { stubComplete, stubEnabled, MARK_SKILL, MARK_BASE };
75
+ module.exports = { stubComplete, stubEnabled, stubUsage, MARK_SKILL, MARK_BASE, STUB_HARNESS_PREAMBLE_TOKENS };
package/lib/usage.js ADDED
@@ -0,0 +1,168 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // Usage capture — normalizing what each surface reports about a single call.
5
+ //
6
+ // Every lane reports token usage in its own shape, and two of them (the CLI
7
+ // lanes) were previously discarding it entirely: the `claude -p` default text
8
+ // output carries no usage block at all, and `codex exec --json` streams usage on
9
+ // stdout that we were routing to /dev/null. Spec v0.4 captures it.
10
+ //
11
+ // ONE normalized shape, so three substrates are comparable:
12
+ //
13
+ // { input_tokens, output_tokens, cached_tokens, wall_ms }
14
+ //
15
+ // input_tokens TOTAL input presented to the model for this call, INCLUDING
16
+ // any portion served from cache. This is the field that most
17
+ // needs normalizing: the surfaces disagree about it (below).
18
+ // cached_tokens the portion of input_tokens served from cache (null when the
19
+ // surface does not surface it — never guessed, never 0-filled).
20
+ // output_tokens tokens generated, INCLUDING reasoning/thinking tokens where
21
+ // the surface bundles them (both CLI lanes do; the split is
22
+ // reported by the surface but deliberately not modelled here —
23
+ // see the disclosure in REPORT-STYLE.md).
24
+ // wall_ms measured BY US around the call (see lib/provider), not taken
25
+ // from the surface. It is the only field whose definition is
26
+ // identical across lanes, which is exactly why latency is
27
+ // reported as an observed, surface-disclosed number.
28
+ //
29
+ // THE INPUT-TOKEN DISAGREEMENT (verified against real output, 2026-08-18):
30
+ // - `claude -p --output-format json` reports input_tokens EXCLUDING cache:
31
+ // total input = input_tokens + cache_creation_input_tokens + cache_read_input_tokens.
32
+ // (Observed: 10 + 6,974 + 18,134 = 25,118 for a two-word prompt.)
33
+ // - `codex exec --json` reports input_tokens INCLUDING cache, with
34
+ // cached_input_tokens as a SUBSET of it. (Observed: 10,807 of which 4,480 cached.)
35
+ // Normalizing to "total including cache" makes the two comparable; the raw
36
+ // fixtures both parsers are tested against are checked in under tests/fixtures/.
37
+ //
38
+ // Both CLI surfaces prepend a large fixed harness preamble that we do not
39
+ // control (~25k tokens on claude-cli, ~11k on codex). It is IDENTICAL in the
40
+ // with-skill and baseline arms, so it cancels in the incremental (Δ) figures —
41
+ // which is why Δcost is the honest axis and absolute per-call cost is not.
42
+
43
+ function n(x) { return Number.isFinite(Number(x)) ? Number(x) : 0; }
44
+
45
+ // The empty/unknown usage record. Deliberately null (not zeros) so "the surface
46
+ // did not tell us" never reads as "the call cost nothing".
47
+ function emptyUsage() {
48
+ return { input_tokens: null, output_tokens: null, cached_tokens: null, wall_ms: null };
49
+ }
50
+
51
+ function hasUsage(u) {
52
+ return !!(u && (u.input_tokens != null || u.output_tokens != null));
53
+ }
54
+
55
+ // ── anthropic/cli — `claude -p --output-format json` ──────────────────────────
56
+ // The whole call is one JSON object; `result` carries the final text and `usage`
57
+ // the token counts. input_tokens EXCLUDES cache, so both cache fields are added
58
+ // back to reach the total actually presented to the model.
59
+ function parseClaudeCliJson(stdout) {
60
+ let j;
61
+ try { j = JSON.parse(String(stdout || '')); } catch (_e) { return null; }
62
+ if (!j || typeof j !== 'object') return null;
63
+ const u = j.usage || {};
64
+ const cacheRead = n(u.cache_read_input_tokens);
65
+ const cacheCreate = n(u.cache_creation_input_tokens);
66
+ const usage = {
67
+ input_tokens: n(u.input_tokens) + cacheRead + cacheCreate,
68
+ output_tokens: n(u.output_tokens),
69
+ // Cache READ is the portion genuinely served from cache. Cache CREATION was
70
+ // processed fresh this call (and billed at a premium), so it is not cached.
71
+ cached_tokens: cacheRead,
72
+ wall_ms: null,
73
+ };
74
+ return {
75
+ text: typeof j.result === 'string' ? j.result : '',
76
+ usage,
77
+ isError: j.is_error === true,
78
+ // The CLI reports its own dollar figure. Recorded here for completeness but
79
+ // NOT used: costs are computed uniformly from the frozen pricing snapshot so
80
+ // three substrates are on one basis (see lib/value.js).
81
+ surfaceReportedCostUsd: Number.isFinite(Number(j.total_cost_usd)) ? Number(j.total_cost_usd) : null,
82
+ };
83
+ }
84
+
85
+ // ── openai/cli — `codex exec --json` JSONL event stream ───────────────────────
86
+ // One JSON object per line. Usage rides the terminal `turn.completed` event;
87
+ // input_tokens already INCLUDES cached_input_tokens. Unparseable lines are
88
+ // skipped (the stream also carries progress events we do not model).
89
+ function parseCodexJsonl(stdout) {
90
+ const lines = String(stdout || '').split('\n').map((l) => l.trim()).filter(Boolean);
91
+ let usage = null;
92
+ let text = '';
93
+ for (const line of lines) {
94
+ let ev;
95
+ try { ev = JSON.parse(line); } catch (_e) { continue; }
96
+ if (!ev || typeof ev !== 'object') continue;
97
+ if (ev.type === 'turn.completed' && ev.usage) {
98
+ const u = ev.usage;
99
+ usage = {
100
+ input_tokens: n(u.input_tokens), // already total (cache included)
101
+ output_tokens: n(u.output_tokens), // includes reasoning_output_tokens
102
+ cached_tokens: u.cached_input_tokens == null ? null : n(u.cached_input_tokens),
103
+ wall_ms: null,
104
+ };
105
+ }
106
+ // The final message is normally read from the -o file; this is a fallback.
107
+ if (ev.type === 'item.completed' && ev.item && ev.item.type === 'agent_message' && typeof ev.item.text === 'string') {
108
+ text = ev.item.text;
109
+ }
110
+ }
111
+ return { usage, text };
112
+ }
113
+
114
+ // ── anthropic/api ─────────────────────────────────────────────────────────────
115
+ // The Messages API reports input_tokens EXCLUDING cache, same as the CLI.
116
+ function parseAnthropicApiUsage(u) {
117
+ if (!u) return emptyUsage();
118
+ const cacheRead = n(u.cache_read_input_tokens);
119
+ const cacheCreate = n(u.cache_creation_input_tokens);
120
+ return {
121
+ input_tokens: n(u.input_tokens) + cacheRead + cacheCreate,
122
+ output_tokens: n(u.output_tokens),
123
+ cached_tokens: u.cache_read_input_tokens == null ? null : cacheRead,
124
+ wall_ms: null,
125
+ };
126
+ }
127
+
128
+ // ── openai/api (Chat Completions-compatible) ──────────────────────────────────
129
+ // prompt_tokens is the total; the cached portion, when present, is nested under
130
+ // prompt_tokens_details.cached_tokens.
131
+ function parseOpenaiApiUsage(u) {
132
+ if (!u) return emptyUsage();
133
+ const details = u.prompt_tokens_details || {};
134
+ return {
135
+ input_tokens: n(u.prompt_tokens),
136
+ output_tokens: n(u.completion_tokens),
137
+ cached_tokens: details.cached_tokens == null ? null : n(details.cached_tokens),
138
+ wall_ms: null,
139
+ };
140
+ }
141
+
142
+ // Sum a list of usage records into one (used for the N judge samples of a case).
143
+ // Null-safe: unknown fields stay null unless at least one record reported them.
144
+ function sumUsage(list) {
145
+ const records = (list || []).filter(Boolean);
146
+ if (!records.length) return emptyUsage();
147
+ const out = { input_tokens: null, output_tokens: null, cached_tokens: null, wall_ms: null };
148
+ for (const key of ['input_tokens', 'output_tokens', 'cached_tokens', 'wall_ms']) {
149
+ const present = records.filter((r) => r[key] != null);
150
+ if (present.length) out[key] = present.reduce((a, r) => a + n(r[key]), 0);
151
+ }
152
+ return out;
153
+ }
154
+
155
+ // Round a usage record's fields to integers (tokens and ms are whole units).
156
+ function normalizeUsage(u) {
157
+ if (!u) return emptyUsage();
158
+ const out = emptyUsage();
159
+ for (const key of ['input_tokens', 'output_tokens', 'cached_tokens', 'wall_ms']) {
160
+ if (u[key] != null && Number.isFinite(Number(u[key]))) out[key] = Math.round(Number(u[key]));
161
+ }
162
+ return out;
163
+ }
164
+
165
+ module.exports = {
166
+ emptyUsage, hasUsage, sumUsage, normalizeUsage,
167
+ parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage,
168
+ };