driftproof 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -162,7 +162,7 @@ A receipt is the unit of evidence — one JSON document conforming to
162
162
 
163
163
  ```jsonc
164
164
  {
165
- "schema_version": "0.3.1",
165
+ "schema_version": "0.4",
166
166
  "skill": { "name": "commit-message-conventions", "version": "0.2.0",
167
167
  "content_hash": "…sha256 over SKILL.md + bundled files…" },
168
168
  "suite": { "format": "agentskills.io/evals", "suite_hash": "…", "case_count": 10 },
@@ -171,7 +171,7 @@ A receipt is the unit of evidence — one JSON document conforming to
171
171
  "model_release_date": "2025-10-01",
172
172
  "provider": "anthropic",
173
173
  "surface": "claude-cli",
174
- "runner_version": "0.4.0",
174
+ "runner_version": "0.5.0",
175
175
  "date_utc": "2026-07-27T…Z",
176
176
  "registry": "registered",
177
177
  "transcripts": "hashes-only",
@@ -233,7 +233,7 @@ jobs:
233
233
  runs-on: ubuntu-latest
234
234
  steps:
235
235
  - uses: actions/checkout@v4
236
- - uses: driftproofhq/driftproof@v0.4.0
236
+ - uses: driftproofhq/driftproof@v0.5.0
237
237
  with:
238
238
  skill-dir: skills/my-skill
239
239
  models: claude-haiku-4-5
package/config.js CHANGED
@@ -9,7 +9,7 @@ const PROJECT_NAME = 'driftproof';
9
9
  // Bumped whenever the runner's behaviour or receipt-generation semantics change
10
10
  // in a way that could affect results. Recorded into every receipt as
11
11
  // run.runner_version so a receipt is reproducible against a known engine.
12
- const RUNNER_VERSION = '0.4.0';
12
+ const RUNNER_VERSION = '0.5.0';
13
13
 
14
14
  // The eval format we CONSUME (we deliberately do not invent our own).
15
15
  const SUITE_FORMAT = 'agentskills.io/evals';
@@ -22,7 +22,14 @@ const SUITE_FORMAT = 'agentskills.io/evals';
22
22
  // surface enums (openai-api/openai-cli), optional run.surface_overhead_note,
23
23
  // optional per-case checks[] (deterministic post-checks), and optional
24
24
  // skill.tokens (value-per-token axis). v0.1/v0.2/v0.3 receipts still load.
25
- const RECEIPT_SCHEMA_VERSION = '0.3.1';
25
+ // v0.4 (additive over v0.3.1) adds the ECONOMICS axis: per-case, per-arm
26
+ // generation `usage` (input/output/cached tokens + measured wall_ms), a
27
+ // separate per-case `judge_usage` (measurement overhead, excluded from
28
+ // every skill-value figure), run.pricing_snapshot (registry prices frozen
29
+ // at run time so derived dollars stay reproducible), and the derived
30
+ // `economics` block. v0.3.1 is frozen as receipt.v0.3.1.schema.json;
31
+ // v0.1/v0.2/v0.3/v0.3.1 receipts all still load.
32
+ const RECEIPT_SCHEMA_VERSION = '0.4';
26
33
 
27
34
  // Hard USD budget defaults per entry point (Week 4). --max-usd overrides any of
28
35
  // these. The projection is refused before any call if it exceeds the cap, on
@@ -73,10 +80,19 @@ const REPORT_004_BASE_MODEL = 'claude-opus-5'; // flagship tier
73
80
  const REPORT_004_FRONTIER_MODEL = 'claude-fable-5'; // frontier tier (full id — no alias)
74
81
  const REPORT_004_JUDGE_MODEL = 'claude-haiku-4-5';
75
82
 
83
+ // Report #005 is a VALUE report — the fourth report type. It asks what a skill
84
+ // COSTS to run alongside whether it helps, over three substrates, and shows the
85
+ // three axes (accuracy lift / Δcost / Δlatency) side by side and never combined.
86
+ // Same suites, same fixed judge; the substrate list spans two providers so the
87
+ // economics are read across surfaces, not within one vendor's pricing.
88
+ const REPORT_005_MODELS = ['claude-sonnet-5', 'claude-fable-5', 'gpt-5.6-sol'];
89
+ const REPORT_005_JUDGE_MODEL = 'claude-haiku-4-5';
90
+
76
91
  module.exports = {
77
92
  PROJECT_NAME, RUNNER_VERSION, SUITE_FORMAT, RECEIPT_SCHEMA_VERSION, DEFAULT_JUDGE_SAMPLES,
78
93
  EFFECT_FLOOR, DEV_MAX_USD, REPORT_MAX_USD, TRIGGER_MAX_USD,
79
94
  REPORT_002_CLAUDE_MODEL, REPORT_002_GPT_MODEL, REPORT_002_JUDGE_MODEL,
80
95
  REPORT_003_NEW_MODEL, REPORT_003_OLD_MODEL, REPORT_003_JUDGE_MODEL,
81
96
  REPORT_004_BASE_MODEL, REPORT_004_FRONTIER_MODEL, REPORT_004_JUDGE_MODEL,
97
+ REPORT_005_MODELS, REPORT_005_JUDGE_MODEL,
82
98
  };
package/lib/judge.js CHANGED
@@ -5,6 +5,7 @@ const { complete, surfaceForModel } = require('./provider');
5
5
  const { extractJsonObject } = require('./json');
6
6
  const { sha256 } = require('./canonical');
7
7
  const { mean, stddev } = require('./stats');
8
+ const { sumUsage } = require('./usage');
8
9
 
9
10
  // Rubric-based LLM judge.
10
11
  //
@@ -82,16 +83,16 @@ function judgeSettings(samples, judgeModel) {
82
83
  // transcript auditability, and optionally retained under --keep-transcripts).
83
84
  async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature }) {
84
85
  const prompt = buildJudgePrompt({ task, response, rubric });
85
- const { text, attempts } = await complete({ system: JUDGE_SYSTEM, prompt, model, maxTokens: 400, timeoutMs, temperature });
86
+ const { text, attempts, usage } = await complete({ system: JUDGE_SYSTEM, prompt, model, maxTokens: 400, timeoutMs, temperature });
86
87
  let parsed;
87
88
  try {
88
89
  parsed = extractJsonObject(text);
89
90
  } catch (_e) {
90
91
  // Unsalvageable judge output → conservative 0 (a judge that can't be parsed
91
92
  // must never silently "pass"), tagged so the caller can see it happened.
92
- return { score: 0, reason: 'judge output unparseable', unparsed: true, raw: String(text || ''), attempts: attempts || 1 };
93
+ return { score: 0, reason: 'judge output unparseable', unparsed: true, raw: String(text || ''), attempts: attempts || 1, usage };
93
94
  }
94
- return { score: clamp01(parsed.score), reason: String(parsed.reason || '').slice(0, 300), raw: String(text || ''), attempts: attempts || 1 };
95
+ return { score: clamp01(parsed.score), reason: String(parsed.reason || '').slice(0, 300), raw: String(text || ''), attempts: attempts || 1, usage };
95
96
  }
96
97
 
97
98
  // Grade a response N times and return the sampled distribution:
@@ -103,6 +104,7 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
103
104
  const scores = [];
104
105
  const reasons = [];
105
106
  const rawTexts = [];
107
+ const usages = [];
106
108
  let attemptsTotal = 0;
107
109
  for (let i = 0; i < samples; i++) {
108
110
  let r;
@@ -116,6 +118,7 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
116
118
  throw e;
117
119
  }
118
120
  attemptsTotal += r.attempts || 1;
121
+ usages.push(r.usage || null);
119
122
  scores.push(r.score);
120
123
  reasons.push(r.reason);
121
124
  rawTexts.push(r.raw || '');
@@ -134,6 +137,11 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
134
137
  model_id: model,
135
138
  rubric_hash: rubricHash(rubric),
136
139
  attempts: attemptsTotal,
140
+ // v0.4: the measurement overhead of grading this one case — the SUM over all
141
+ // N judge calls. Recorded in the receipt as the case's `judge_usage` and
142
+ // EXCLUDED from every skill-value figure (lib/value.js): it is a cost we
143
+ // impose to measure, not a cost of running the skill.
144
+ usage: sumUsage(usages),
137
145
  };
138
146
  }
139
147
 
package/lib/provider.js CHANGED
@@ -7,6 +7,9 @@ const path = require('path');
7
7
  const { spawn } = require('child_process');
8
8
  const { withRetry, withTimeout } = require('./json');
9
9
  const { stubComplete, stubEnabled } = require('./stub');
10
+ const {
11
+ parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage, normalizeUsage,
12
+ } = require('./usage');
10
13
 
11
14
  // Provider abstraction: one `complete()` call over a TWO-AXIS lane model —
12
15
  // provider (anthropic | openai) × surface (api | cli). Four concrete lanes:
@@ -109,9 +112,18 @@ const CODEX_OVERHEAD_NOTE =
109
112
  + 'control; the model id is set by us via -m (it is not echoed in the JSONL stream). '
110
113
  + 'Approval prompting is off by default on `codex exec` (no -a flag is passed).';
111
114
 
112
- // Send a single-turn prompt and return { text, usage, surface, provider }.
113
- // usage is { input_tokens, output_tokens } when the surface reports it (api
114
- // surfaces), else null (cli surfaces do not expose token counts to us).
115
+ // Send a single-turn prompt and return { text, usage, wall_ms, surface, provider }.
116
+ //
117
+ // usage is the normalized v0.4 record { input_tokens, output_tokens, cached_tokens,
118
+ // wall_ms } on EVERY lane — including the two CLI lanes, which report it in their
119
+ // structured output (`claude -p --output-format json`; the `codex exec --json`
120
+ // JSONL stream). See lib/usage.js for the per-surface shapes and the input-token
121
+ // normalization. Fields the surface does not report stay null, never 0.
122
+ //
123
+ // wall_ms is measured HERE, around the successful attempt, so it means the same
124
+ // thing on all four lanes (retries are excluded — a retried call's latency would
125
+ // describe our backoff, not the model).
126
+ //
115
127
  // `temperature` is honoured ONLY on api surfaces; cli surfaces control sampling
116
128
  // themselves, so temperature is ignored there and the receipt records that fact.
117
129
  async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = null, temperature = undefined }) {
@@ -120,7 +132,10 @@ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = n
120
132
  // Offline stub surface: canned completion, zero model calls. The receipt still
121
133
  // records the real surface/provider so a stub run is not mistaken for a genuine
122
134
  // one at read time — only the generation/judge TEXT is canned.
123
- if (stubEnabled()) return { ...stubComplete({ system, prompt }), surface, provider, attempts: 1 };
135
+ if (stubEnabled()) {
136
+ const s = stubComplete({ system, prompt });
137
+ return { ...s, usage: normalizeUsage(s.usage), wall_ms: (s.usage && s.usage.wall_ms) || 0, surface, provider, attempts: 1 };
138
+ }
124
139
 
125
140
  // Per-surface tolerance (see retryPolicyForSurface): a cli/subscription surface
126
141
  // spawns a first-party CLI subprocess whose cold-start is slow and occasionally
@@ -144,10 +159,20 @@ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = n
144
159
  // throttling/cold-starts so one blip doesn't abort a multi-hour grind. `attempts`
145
160
  // counts every try (retries included) so the budget can charge for them.
146
161
  let attempts = 0;
147
- const runner = () => { attempts += 1; return withTimeout(laneRunner, effTimeout, `provider(${surface})`); };
162
+ // Wall-clock of the attempt that SUCCEEDED (each attempt overwrites, so a
163
+ // retried call reports the latency of the call that actually produced the text,
164
+ // not the accumulated backoff).
165
+ let wallMs = null;
166
+ const runner = () => {
167
+ attempts += 1;
168
+ const t0 = Date.now();
169
+ return withTimeout(laneRunner, effTimeout, `provider(${surface})`)
170
+ .then((r) => { wallMs = Date.now() - t0; return r; });
171
+ };
148
172
  try {
149
173
  const out = await withRetry(runner, { tries: policy.tries, baseDelayMs });
150
- return { ...out, surface, provider, attempts };
174
+ const usage = normalizeUsage({ ...(out.usage || {}), wall_ms: wallMs });
175
+ return { ...out, usage, wall_ms: wallMs, surface, provider, attempts };
151
176
  } catch (e) {
152
177
  // Surface the attempt count so a persistently-failing call can be charged for
153
178
  // (and, for a timeout, marked failed_timeout by the runner instead of fatal).
@@ -172,13 +197,7 @@ async function completeAnthropicApi({ system, prompt, model, maxTokens, temperat
172
197
  if (temperature !== undefined) params.temperature = temperature;
173
198
  const resp = await client.messages.create(params);
174
199
  const text = (resp.content || []).filter((b) => b.type === 'text').map((b) => b.text).join('');
175
- return {
176
- text,
177
- usage: {
178
- input_tokens: (resp.usage && resp.usage.input_tokens) || 0,
179
- output_tokens: (resp.usage && resp.usage.output_tokens) || 0,
180
- },
181
- };
200
+ return { text, usage: parseAnthropicApiUsage(resp.usage) };
182
201
  }
183
202
 
184
203
  // ── anthropic/cli (claude -p) ──────────────────────────────────────────────────
@@ -189,7 +208,13 @@ function completeClaudeCli({ system, prompt, model, timeoutMs }) {
189
208
  const env = { ...process.env };
190
209
  delete env.ANTHROPIC_API_KEY;
191
210
 
192
- const args = ['-p', '--model', resolveModel(model)];
211
+ // v0.4: `--output-format json` returns ONE JSON object carrying both the
212
+ // final text (`result`) and the token usage (`usage`) — the default text
213
+ // output carries no usage at all, which is why usage was previously null on
214
+ // this lane. The text is read from the parsed object; if the CLI ever emits
215
+ // something unparseable we fall back to the raw stdout so a run degrades to
216
+ // the old behaviour (text, no usage) rather than failing.
217
+ const args = ['-p', '--output-format', 'json', '--model', resolveModel(model)];
193
218
  if (system) args.push('--append-system-prompt', system);
194
219
 
195
220
  const child = spawn('claude', args, { env, stdio: ['pipe', 'pipe', 'pipe'] });
@@ -206,7 +231,10 @@ function completeClaudeCli({ system, prompt, model, timeoutMs }) {
206
231
  child.on('close', (code) => {
207
232
  clearTimeout(killer);
208
233
  if (code !== 0) return reject(new Error(`claude CLI exited ${code}: ${err.slice(0, 400)}`));
209
- resolve({ text: out.trim(), usage: null });
234
+ const parsed = parseClaudeCliJson(out);
235
+ if (!parsed) return resolve({ text: out.trim(), usage: null });
236
+ if (parsed.isError) return reject(new Error(`claude CLI reported is_error: ${String(parsed.text || '').slice(0, 300)}`));
237
+ resolve({ text: String(parsed.text || '').trim(), usage: parsed.usage });
210
238
  });
211
239
  child.stdin.write(prompt);
212
240
  child.stdin.end();
@@ -250,14 +278,7 @@ async function completeOpenaiApi({ system, prompt, model, maxTokens, temperature
250
278
  let parsed;
251
279
  try { parsed = JSON.parse(raw); } catch (_e) { throw new Error(`openai/api: unparseable response: ${raw.slice(0, 200)}`); }
252
280
  const text = ((parsed.choices && parsed.choices[0] && parsed.choices[0].message && parsed.choices[0].message.content) || '');
253
- const usage = parsed.usage || {};
254
- return {
255
- text: String(text).trim(),
256
- usage: {
257
- input_tokens: usage.prompt_tokens || 0,
258
- output_tokens: usage.completion_tokens || 0,
259
- },
260
- };
281
+ return { text: String(text).trim(), usage: parseOpenaiApiUsage(parsed.usage) };
261
282
  }
262
283
 
263
284
  // Read the OpenAI provider config (base_url + api-key env) from the registry,
@@ -325,9 +346,15 @@ function completeCodexCli({ system, prompt, model, timeoutMs }) {
325
346
  // stdin (per its --help), which is content-agnostic.
326
347
  const args = codexFinalArgs({ model, outFile });
327
348
 
328
- const child = spawn('codex', args, { env: { ...process.env }, stdio: ['pipe', 'ignore', 'pipe'] });
349
+ // v0.4: stdout is CAPTURED (it was 'ignore'). `--json` streams JSONL events
350
+ // there, and the terminal `turn.completed` event carries this call's token
351
+ // usage — the only place codex reports it. The final message still comes from
352
+ // the -o file (cleaner than scraping the stream); the JSONL is read for usage.
353
+ const child = spawn('codex', args, { env: { ...process.env }, stdio: ['pipe', 'pipe', 'pipe'] });
329
354
  let err = '';
355
+ let jsonl = '';
330
356
  const killer = setTimeout(() => { child.kill('SIGKILL'); }, timeoutMs + 5000);
357
+ child.stdout.on('data', (d) => { jsonl += d; });
331
358
  child.stderr.on('data', (d) => { err += d; });
332
359
  child.on('error', (e) => {
333
360
  clearTimeout(killer);
@@ -340,7 +367,10 @@ function completeCodexCli({ system, prompt, model, timeoutMs }) {
340
367
  try { text = fs.readFileSync(outFile, 'utf8'); } catch (_e) { /* fall through */ }
341
368
  try { fs.unlinkSync(outFile); } catch (_e) { /* best effort */ }
342
369
  if (code !== 0) return reject(new Error(`codex exec exited ${code}: ${String(err).slice(0, 400)}`));
343
- resolve({ text: String(text).trim(), usage: null });
370
+ const ev = parseCodexJsonl(jsonl);
371
+ // Prefer the -o file; fall back to the stream's agent_message if it is empty.
372
+ const finalText = String(text || ev.text || '').trim();
373
+ resolve({ text: finalText, usage: ev.usage });
344
374
  });
345
375
  child.stdin.on('error', () => { /* ignore EPIPE if codex exits before reading all input */ });
346
376
  child.stdin.write(fullPrompt);
package/lib/receipt.js CHANGED
@@ -14,7 +14,8 @@ const SCHEMA_FILES = {
14
14
  '0.1': 'receipt.v0.1.schema.json',
15
15
  '0.2': 'receipt.v0.2.schema.json',
16
16
  '0.3': 'receipt.v0.3.schema.json',
17
- '0.3.1': 'receipt.schema.json',
17
+ '0.3.1': 'receipt.v0.3.1.schema.json',
18
+ '0.4': 'receipt.schema.json',
18
19
  };
19
20
 
20
21
  const _validators = {};
@@ -80,7 +81,7 @@ function aggregate(caseResults) {
80
81
  // run: { model_id, model_release_date, surface, runner_version, date_utc, judge, registry, transcripts }
81
82
  // cases: [ { id, mode, outcome, score, mean, stddev, samples, generation_hash, judge_sample_hashes, threshold, reason, judge } ]
82
83
  // editorialReviews: optional [ { url, source, date } ]
83
- function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED', editorialReviews = null }) {
84
+ function buildReceipt({ skill, suite, run, cases, economics = null, verificationLevel = 'TESTED', editorialReviews = null }) {
84
85
  // v0.3.1: cases marked failed_timeout are recorded in results.cases but EXCLUDED
85
86
  // from aggregates — a band is never fabricated from a case that did not complete.
86
87
  const okCases = cases.filter((c) => c.case_status !== 'failed_timeout');
@@ -138,6 +139,11 @@ function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED',
138
139
  // skill.tokens — estimated SKILL.md token size (value-per-token axis).
139
140
  if (run.surface_overhead_note) receipt.run.surface_overhead_note = run.surface_overhead_note;
140
141
  if (skill.tokens != null) receipt.skill.tokens = skill.tokens;
142
+ // v0.4 economics (additive-optional): the frozen prices this receipt's derived
143
+ // dollar figures were computed from, and the derived block itself. A receipt
144
+ // from a surface that reports no usage simply omits both.
145
+ if (run.pricing_snapshot) receipt.run.pricing_snapshot = run.pricing_snapshot;
146
+ if (economics) receipt.economics = economics;
141
147
  // v0.3.1: mark the receipt incomplete when any case failed (excluded above).
142
148
  if (failedCount > 0) {
143
149
  receipt.run.status = 'incomplete';
package/lib/run.js CHANGED
@@ -1,14 +1,16 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
- const { complete, resolveModel, surfaceForModel, CODEX_OVERHEAD_NOTE } = require('./provider');
4
+ const { complete, resolveModel, surfaceForModel, isMeteredSurface, CODEX_OVERHEAD_NOTE } = require('./provider');
5
5
  const { gradeSamples, judgeSettings } = require('./judge');
6
6
  const { buildReceipt } = require('./receipt');
7
7
  const { sha256 } = require('./canonical');
8
- const { registryStatus, providerForModel } = require('./models');
8
+ const { registryStatus, providerForModel, priceForModel } = require('./models');
9
9
  const { perCallCostUSD } = require('./cost');
10
10
  const { runChecks } = require('./checks');
11
11
  const { estimateTokens } = require('./skillCost');
12
+ const { hasUsage, normalizeUsage } = require('./usage');
13
+ const { buildPricingSnapshot, computeEconomics } = require('./value');
12
14
  const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES } = require('../config');
13
15
 
14
16
  // Known model release dates (best-effort; null when unknown). Recorded into the
@@ -51,8 +53,8 @@ async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs }) {
51
53
  throw e;
52
54
  }
53
55
  const system = withSkill ? skillMd : undefined;
54
- const { text, usage, attempts } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
55
- return { text, usage, attempts };
56
+ const { text, usage, wall_ms, attempts } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
57
+ return { text, usage, wall_ms, attempts };
56
58
  }
57
59
 
58
60
  // Determine a case outcome from its sampled band and threshold.
@@ -91,6 +93,8 @@ async function judgeCase({ caseObj, response, generationHash, judgeModel, mode,
91
93
  // v0.3.1 deterministic post-checks (supplementary; NOT folded into `outcome`).
92
94
  const checks = runChecks(response, caseObj.checks);
93
95
  if (checks.length) caseResult.checks = checks;
96
+ // v0.4: grading overhead for this case row, kept OUT of the skill-value math.
97
+ if (hasUsage(g.usage)) caseResult.judge_usage = g.usage;
94
98
  return { caseResult, sampleTexts: g.sample_texts, attempts: g.attempts };
95
99
  }
96
100
 
@@ -170,6 +174,9 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
170
174
  const generationHash = sha256(String(gen.text || ''));
171
175
  onProgress({ case: c.id, mode, phase: 'judge', samples });
172
176
  const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ct, samples });
177
+ // v0.4: the GENERATION call's usage is the skill-value measurement (the
178
+ // judge's own usage rides separately on judge_usage, above).
179
+ if (hasUsage(gen.usage)) jr.caseResult.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
173
180
  calls += samples;
174
181
  // Live budget: count all judge calls (retries included) for this (case, mode).
175
182
  if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judgeModel, 'judge'));
@@ -198,6 +205,24 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
198
205
  const transcripts = keepTranscripts ? pairs.map((p) => p.transcript) : null;
199
206
 
200
207
  const surface = surfaceForModel(modelId);
208
+ const nowIso = opts.nowIso || new Date().toISOString();
209
+ // v0.4 economics. The pricing snapshot is frozen HERE, at run time, from the
210
+ // registry; every derived dollar figure below is computed from the snapshot and
211
+ // never from the live registry, so this receipt keeps its meaning when prices
212
+ // later change.
213
+ const pricingSnapshot = buildPricingSnapshot({
214
+ models: [modelId, judgeModel],
215
+ lookup: priceForModel,
216
+ nowIso,
217
+ });
218
+ const economics = computeEconomics({
219
+ cases: caseResults,
220
+ modelId,
221
+ judgeModelId: judgeModel,
222
+ pricingSnapshot,
223
+ surface,
224
+ meteredSurface: isMeteredSurface(surface),
225
+ });
201
226
  const receipt = buildReceipt({
202
227
  skill: {
203
228
  name: skill.name, version: skill.version, contentHash: skill.contentHash,
@@ -213,12 +238,14 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
213
238
  // v0.3.1: on the openai/cli (codex) surface, record the fixed harness preamble.
214
239
  surface_overhead_note: surface === 'openai-cli' ? CODEX_OVERHEAD_NOTE : undefined,
215
240
  runner_version: RUNNER_VERSION,
216
- date_utc: opts.nowIso || new Date().toISOString(),
241
+ date_utc: nowIso,
217
242
  registry: registryStatus(modelId),
218
243
  transcripts: keepTranscripts ? 'retained-local' : 'hashes-only',
219
244
  judge: judgeSettings(samples, judgeModel),
245
+ pricing_snapshot: pricingSnapshot,
220
246
  },
221
247
  cases: caseResults,
248
+ economics,
222
249
  verificationLevel: 'TESTED',
223
250
  });
224
251
 
package/lib/stub.js CHANGED
@@ -39,16 +39,37 @@ function stubComplete({ system, prompt }) {
39
39
  pass: score >= 0.7,
40
40
  reason: `stub judge (DRIFTPROOF_STUB): ${helped ? 'with-skill marker present' : 'baseline'}`,
41
41
  };
42
- return { text: JSON.stringify(body), usage: null };
42
+ return { text: JSON.stringify(body), usage: stubUsage({ kind: 'judge', prompt }) };
43
43
  }
44
44
  // Generation call: canned, marked by mode.
45
45
  const mark = system ? MARK_SKILL : MARK_BASE;
46
46
  const text = `${mark}\nfeat(stub): canned offline generation for CI (no model was called)`;
47
- return { text, usage: null };
47
+ return { text, usage: stubUsage({ kind: 'gen', system, prompt }) };
48
+ }
49
+
50
+ // v0.4: deterministic synthetic usage, so a zero-model-call stub run still
51
+ // exercises the whole economics path (usage → pricing snapshot → derived cost/
52
+ // latency fields → the value report's three axes). The numbers are SYNTHETIC and
53
+ // deliberately shaped like the real surfaces: a large fixed harness preamble that
54
+ // is identical in both arms, plus the with-skill arm's extra SKILL.md input and
55
+ // its slightly longer, slower output. No randomness — same input, same usage.
56
+ const STUB_HARNESS_PREAMBLE_TOKENS = 20000; // the fixed, arm-identical overhead
57
+ function stubUsage({ kind, system, prompt }) {
58
+ const promptTokens = Math.ceil(String(prompt || '').length / 4);
59
+ if (kind === 'judge') {
60
+ return { input_tokens: 1200 + promptTokens, output_tokens: 60, cached_tokens: 800, wall_ms: 900 };
61
+ }
62
+ const skillTokens = system ? Math.ceil(String(system).length / 4) : 0;
63
+ return {
64
+ input_tokens: STUB_HARNESS_PREAMBLE_TOKENS + promptTokens + skillTokens,
65
+ output_tokens: system ? 420 : 350,
66
+ cached_tokens: STUB_HARNESS_PREAMBLE_TOKENS,
67
+ wall_ms: system ? 4200 : 3600,
68
+ };
48
69
  }
49
70
 
50
71
  function stubEnabled() {
51
72
  return process.env.DRIFTPROOF_STUB === '1';
52
73
  }
53
74
 
54
- module.exports = { stubComplete, stubEnabled, MARK_SKILL, MARK_BASE };
75
+ module.exports = { stubComplete, stubEnabled, stubUsage, MARK_SKILL, MARK_BASE, STUB_HARNESS_PREAMBLE_TOKENS };
package/lib/usage.js ADDED
@@ -0,0 +1,168 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // Usage capture — normalizing what each surface reports about a single call.
5
+ //
6
+ // Every lane reports token usage in its own shape, and two of them (the CLI
7
+ // lanes) were previously discarding it entirely: the `claude -p` default text
8
+ // output carries no usage block at all, and `codex exec --json` streams usage on
9
+ // stdout that we were routing to /dev/null. Spec v0.4 captures it.
10
+ //
11
+ // ONE normalized shape, so three substrates are comparable:
12
+ //
13
+ // { input_tokens, output_tokens, cached_tokens, wall_ms }
14
+ //
15
+ // input_tokens TOTAL input presented to the model for this call, INCLUDING
16
+ // any portion served from cache. This is the field that most
17
+ // needs normalizing: the surfaces disagree about it (below).
18
+ // cached_tokens the portion of input_tokens served from cache (null when the
19
+ // surface does not surface it — never guessed, never 0-filled).
20
+ // output_tokens tokens generated, INCLUDING reasoning/thinking tokens where
21
+ // the surface bundles them (both CLI lanes do; the split is
22
+ // reported by the surface but deliberately not modelled here —
23
+ // see the disclosure in REPORT-STYLE.md).
24
+ // wall_ms measured BY US around the call (see lib/provider), not taken
25
+ // from the surface. It is the only field whose definition is
26
+ // identical across lanes, which is exactly why latency is
27
+ // reported as an observed, surface-disclosed number.
28
+ //
29
+ // THE INPUT-TOKEN DISAGREEMENT (verified against real output, 2026-08-18):
30
+ // - `claude -p --output-format json` reports input_tokens EXCLUDING cache:
31
+ // total input = input_tokens + cache_creation_input_tokens + cache_read_input_tokens.
32
+ // (Observed: 10 + 6,974 + 18,134 = 25,118 for a two-word prompt.)
33
+ // - `codex exec --json` reports input_tokens INCLUDING cache, with
34
+ // cached_input_tokens as a SUBSET of it. (Observed: 10,807 of which 4,480 cached.)
35
+ // Normalizing to "total including cache" makes the two comparable; the raw
36
+ // fixtures both parsers are tested against are checked in under tests/fixtures/.
37
+ //
38
+ // Both CLI surfaces prepend a large fixed harness preamble that we do not
39
+ // control (~25k tokens on claude-cli, ~11k on codex). It is IDENTICAL in the
40
+ // with-skill and baseline arms, so it cancels in the incremental (Δ) figures —
41
+ // which is why Δcost is the honest axis and absolute per-call cost is not.
42
+
43
+ function n(x) { return Number.isFinite(Number(x)) ? Number(x) : 0; }
44
+
45
+ // The empty/unknown usage record. Deliberately null (not zeros) so "the surface
46
+ // did not tell us" never reads as "the call cost nothing".
47
+ function emptyUsage() {
48
+ return { input_tokens: null, output_tokens: null, cached_tokens: null, wall_ms: null };
49
+ }
50
+
51
+ function hasUsage(u) {
52
+ return !!(u && (u.input_tokens != null || u.output_tokens != null));
53
+ }
54
+
55
+ // ── anthropic/cli — `claude -p --output-format json` ──────────────────────────
56
+ // The whole call is one JSON object; `result` carries the final text and `usage`
57
+ // the token counts. input_tokens EXCLUDES cache, so both cache fields are added
58
+ // back to reach the total actually presented to the model.
59
+ function parseClaudeCliJson(stdout) {
60
+ let j;
61
+ try { j = JSON.parse(String(stdout || '')); } catch (_e) { return null; }
62
+ if (!j || typeof j !== 'object') return null;
63
+ const u = j.usage || {};
64
+ const cacheRead = n(u.cache_read_input_tokens);
65
+ const cacheCreate = n(u.cache_creation_input_tokens);
66
+ const usage = {
67
+ input_tokens: n(u.input_tokens) + cacheRead + cacheCreate,
68
+ output_tokens: n(u.output_tokens),
69
+ // Cache READ is the portion genuinely served from cache. Cache CREATION was
70
+ // processed fresh this call (and billed at a premium), so it is not cached.
71
+ cached_tokens: cacheRead,
72
+ wall_ms: null,
73
+ };
74
+ return {
75
+ text: typeof j.result === 'string' ? j.result : '',
76
+ usage,
77
+ isError: j.is_error === true,
78
+ // The CLI reports its own dollar figure. Recorded here for completeness but
79
+ // NOT used: costs are computed uniformly from the frozen pricing snapshot so
80
+ // three substrates are on one basis (see lib/value.js).
81
+ surfaceReportedCostUsd: Number.isFinite(Number(j.total_cost_usd)) ? Number(j.total_cost_usd) : null,
82
+ };
83
+ }
84
+
85
+ // ── openai/cli — `codex exec --json` JSONL event stream ───────────────────────
86
+ // One JSON object per line. Usage rides the terminal `turn.completed` event;
87
+ // input_tokens already INCLUDES cached_input_tokens. Unparseable lines are
88
+ // skipped (the stream also carries progress events we do not model).
89
+ function parseCodexJsonl(stdout) {
90
+ const lines = String(stdout || '').split('\n').map((l) => l.trim()).filter(Boolean);
91
+ let usage = null;
92
+ let text = '';
93
+ for (const line of lines) {
94
+ let ev;
95
+ try { ev = JSON.parse(line); } catch (_e) { continue; }
96
+ if (!ev || typeof ev !== 'object') continue;
97
+ if (ev.type === 'turn.completed' && ev.usage) {
98
+ const u = ev.usage;
99
+ usage = {
100
+ input_tokens: n(u.input_tokens), // already total (cache included)
101
+ output_tokens: n(u.output_tokens), // includes reasoning_output_tokens
102
+ cached_tokens: u.cached_input_tokens == null ? null : n(u.cached_input_tokens),
103
+ wall_ms: null,
104
+ };
105
+ }
106
+ // The final message is normally read from the -o file; this is a fallback.
107
+ if (ev.type === 'item.completed' && ev.item && ev.item.type === 'agent_message' && typeof ev.item.text === 'string') {
108
+ text = ev.item.text;
109
+ }
110
+ }
111
+ return { usage, text };
112
+ }
113
+
114
+ // ── anthropic/api ─────────────────────────────────────────────────────────────
115
+ // The Messages API reports input_tokens EXCLUDING cache, same as the CLI.
116
+ function parseAnthropicApiUsage(u) {
117
+ if (!u) return emptyUsage();
118
+ const cacheRead = n(u.cache_read_input_tokens);
119
+ const cacheCreate = n(u.cache_creation_input_tokens);
120
+ return {
121
+ input_tokens: n(u.input_tokens) + cacheRead + cacheCreate,
122
+ output_tokens: n(u.output_tokens),
123
+ cached_tokens: u.cache_read_input_tokens == null ? null : cacheRead,
124
+ wall_ms: null,
125
+ };
126
+ }
127
+
128
+ // ── openai/api (Chat Completions-compatible) ──────────────────────────────────
129
+ // prompt_tokens is the total; the cached portion, when present, is nested under
130
+ // prompt_tokens_details.cached_tokens.
131
+ function parseOpenaiApiUsage(u) {
132
+ if (!u) return emptyUsage();
133
+ const details = u.prompt_tokens_details || {};
134
+ return {
135
+ input_tokens: n(u.prompt_tokens),
136
+ output_tokens: n(u.completion_tokens),
137
+ cached_tokens: details.cached_tokens == null ? null : n(details.cached_tokens),
138
+ wall_ms: null,
139
+ };
140
+ }
141
+
142
+ // Sum a list of usage records into one (used for the N judge samples of a case).
143
+ // Null-safe: unknown fields stay null unless at least one record reported them.
144
+ function sumUsage(list) {
145
+ const records = (list || []).filter(Boolean);
146
+ if (!records.length) return emptyUsage();
147
+ const out = { input_tokens: null, output_tokens: null, cached_tokens: null, wall_ms: null };
148
+ for (const key of ['input_tokens', 'output_tokens', 'cached_tokens', 'wall_ms']) {
149
+ const present = records.filter((r) => r[key] != null);
150
+ if (present.length) out[key] = present.reduce((a, r) => a + n(r[key]), 0);
151
+ }
152
+ return out;
153
+ }
154
+
155
+ // Round a usage record's fields to integers (tokens and ms are whole units).
156
+ function normalizeUsage(u) {
157
+ if (!u) return emptyUsage();
158
+ const out = emptyUsage();
159
+ for (const key of ['input_tokens', 'output_tokens', 'cached_tokens', 'wall_ms']) {
160
+ if (u[key] != null && Number.isFinite(Number(u[key]))) out[key] = Math.round(Number(u[key]));
161
+ }
162
+ return out;
163
+ }
164
+
165
+ module.exports = {
166
+ emptyUsage, hasUsage, sumUsage, normalizeUsage,
167
+ parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage,
168
+ };