driftproof 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/provider.js CHANGED
@@ -7,6 +7,9 @@ const path = require('path');
7
7
  const { spawn } = require('child_process');
8
8
  const { withRetry, withTimeout } = require('./json');
9
9
  const { stubComplete, stubEnabled } = require('./stub');
10
+ const {
11
+ parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage, normalizeUsage,
12
+ } = require('./usage');
10
13
 
11
14
  // Provider abstraction: one `complete()` call over a TWO-AXIS lane model —
12
15
  // provider (anthropic | openai) × surface (api | cli). Four concrete lanes:
@@ -109,9 +112,18 @@ const CODEX_OVERHEAD_NOTE =
109
112
  + 'control; the model id is set by us via -m (it is not echoed in the JSONL stream). '
110
113
  + 'Approval prompting is off by default on `codex exec` (no -a flag is passed).';
111
114
 
112
- // Send a single-turn prompt and return { text, usage, surface, provider }.
113
- // usage is { input_tokens, output_tokens } when the surface reports it (api
114
- // surfaces), else null (cli surfaces do not expose token counts to us).
115
+ // Send a single-turn prompt and return { text, usage, wall_ms, surface, provider }.
116
+ //
117
+ // usage is the normalized v0.4 record { input_tokens, output_tokens, cached_tokens,
118
+ // wall_ms } on EVERY lane — including the two CLI lanes, which report it in their
119
+ // structured output (`claude -p --output-format json`; the `codex exec --json`
120
+ // JSONL stream). See lib/usage.js for the per-surface shapes and the input-token
121
+ // normalization. Fields the surface does not report stay null, never 0.
122
+ //
123
+ // wall_ms is measured HERE, around the successful attempt, so it means the same
124
+ // thing on all four lanes (retries are excluded — a retried call's latency would
125
+ // describe our backoff, not the model).
126
+ //
115
127
  // `temperature` is honoured ONLY on api surfaces; cli surfaces control sampling
116
128
  // themselves, so temperature is ignored there and the receipt records that fact.
117
129
  async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = null, temperature = undefined }) {
@@ -120,7 +132,10 @@ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = n
120
132
  // Offline stub surface: canned completion, zero model calls. The receipt still
121
133
  // records the real surface/provider so a stub run is not mistaken for a genuine
122
134
  // one at read time — only the generation/judge TEXT is canned.
123
- if (stubEnabled()) return { ...stubComplete({ system, prompt }), surface, provider, attempts: 1 };
135
+ if (stubEnabled()) {
136
+ const s = stubComplete({ system, prompt });
137
+ return { ...s, usage: normalizeUsage(s.usage), wall_ms: (s.usage && s.usage.wall_ms) || 0, surface, provider, attempts: 1 };
138
+ }
124
139
 
125
140
  // Per-surface tolerance (see retryPolicyForSurface): a cli/subscription surface
126
141
  // spawns a first-party CLI subprocess whose cold-start is slow and occasionally
@@ -144,10 +159,20 @@ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = n
144
159
  // throttling/cold-starts so one blip doesn't abort a multi-hour grind. `attempts`
145
160
  // counts every try (retries included) so the budget can charge for them.
146
161
  let attempts = 0;
147
- const runner = () => { attempts += 1; return withTimeout(laneRunner, effTimeout, `provider(${surface})`); };
162
+ // Wall-clock of the attempt that SUCCEEDED (each attempt overwrites, so a
163
+ // retried call reports the latency of the call that actually produced the text,
164
+ // not the accumulated backoff).
165
+ let wallMs = null;
166
+ const runner = () => {
167
+ attempts += 1;
168
+ const t0 = Date.now();
169
+ return withTimeout(laneRunner, effTimeout, `provider(${surface})`)
170
+ .then((r) => { wallMs = Date.now() - t0; return r; });
171
+ };
148
172
  try {
149
173
  const out = await withRetry(runner, { tries: policy.tries, baseDelayMs });
150
- return { ...out, surface, provider, attempts };
174
+ const usage = normalizeUsage({ ...(out.usage || {}), wall_ms: wallMs });
175
+ return { ...out, usage, wall_ms: wallMs, surface, provider, attempts };
151
176
  } catch (e) {
152
177
  // Surface the attempt count so a persistently-failing call can be charged for
153
178
  // (and, for a timeout, marked failed_timeout by the runner instead of fatal).
@@ -172,13 +197,7 @@ async function completeAnthropicApi({ system, prompt, model, maxTokens, temperat
172
197
  if (temperature !== undefined) params.temperature = temperature;
173
198
  const resp = await client.messages.create(params);
174
199
  const text = (resp.content || []).filter((b) => b.type === 'text').map((b) => b.text).join('');
175
- return {
176
- text,
177
- usage: {
178
- input_tokens: (resp.usage && resp.usage.input_tokens) || 0,
179
- output_tokens: (resp.usage && resp.usage.output_tokens) || 0,
180
- },
181
- };
200
+ return { text, usage: parseAnthropicApiUsage(resp.usage) };
182
201
  }
183
202
 
184
203
  // ── anthropic/cli (claude -p) ──────────────────────────────────────────────────
@@ -189,7 +208,13 @@ function completeClaudeCli({ system, prompt, model, timeoutMs }) {
189
208
  const env = { ...process.env };
190
209
  delete env.ANTHROPIC_API_KEY;
191
210
 
192
- const args = ['-p', '--model', resolveModel(model)];
211
+ // v0.4: `--output-format json` returns ONE JSON object carrying both the
212
+ // final text (`result`) and the token usage (`usage`) — the default text
213
+ // output carries no usage at all, which is why usage was previously null on
214
+ // this lane. The text is read from the parsed object; if the CLI ever emits
215
+ // something unparseable we fall back to the raw stdout so a run degrades to
216
+ // the old behaviour (text, no usage) rather than failing.
217
+ const args = ['-p', '--output-format', 'json', '--model', resolveModel(model)];
193
218
  if (system) args.push('--append-system-prompt', system);
194
219
 
195
220
  const child = spawn('claude', args, { env, stdio: ['pipe', 'pipe', 'pipe'] });
@@ -206,7 +231,10 @@ function completeClaudeCli({ system, prompt, model, timeoutMs }) {
206
231
  child.on('close', (code) => {
207
232
  clearTimeout(killer);
208
233
  if (code !== 0) return reject(new Error(`claude CLI exited ${code}: ${err.slice(0, 400)}`));
209
- resolve({ text: out.trim(), usage: null });
234
+ const parsed = parseClaudeCliJson(out);
235
+ if (!parsed) return resolve({ text: out.trim(), usage: null });
236
+ if (parsed.isError) return reject(new Error(`claude CLI reported is_error: ${String(parsed.text || '').slice(0, 300)}`));
237
+ resolve({ text: String(parsed.text || '').trim(), usage: parsed.usage });
210
238
  });
211
239
  child.stdin.write(prompt);
212
240
  child.stdin.end();
@@ -250,14 +278,7 @@ async function completeOpenaiApi({ system, prompt, model, maxTokens, temperature
250
278
  let parsed;
251
279
  try { parsed = JSON.parse(raw); } catch (_e) { throw new Error(`openai/api: unparseable response: ${raw.slice(0, 200)}`); }
252
280
  const text = ((parsed.choices && parsed.choices[0] && parsed.choices[0].message && parsed.choices[0].message.content) || '');
253
- const usage = parsed.usage || {};
254
- return {
255
- text: String(text).trim(),
256
- usage: {
257
- input_tokens: usage.prompt_tokens || 0,
258
- output_tokens: usage.completion_tokens || 0,
259
- },
260
- };
281
+ return { text: String(text).trim(), usage: parseOpenaiApiUsage(parsed.usage) };
261
282
  }
262
283
 
263
284
  // Read the OpenAI provider config (base_url + api-key env) from the registry,
@@ -325,9 +346,15 @@ function completeCodexCli({ system, prompt, model, timeoutMs }) {
325
346
  // stdin (per its --help), which is content-agnostic.
326
347
  const args = codexFinalArgs({ model, outFile });
327
348
 
328
- const child = spawn('codex', args, { env: { ...process.env }, stdio: ['pipe', 'ignore', 'pipe'] });
349
+ // v0.4: stdout is CAPTURED (it was 'ignore'). `--json` streams JSONL events
350
+ // there, and the terminal `turn.completed` event carries this call's token
351
+ // usage — the only place codex reports it. The final message still comes from
352
+ // the -o file (cleaner than scraping the stream); the JSONL is read for usage.
353
+ const child = spawn('codex', args, { env: { ...process.env }, stdio: ['pipe', 'pipe', 'pipe'] });
329
354
  let err = '';
355
+ let jsonl = '';
330
356
  const killer = setTimeout(() => { child.kill('SIGKILL'); }, timeoutMs + 5000);
357
+ child.stdout.on('data', (d) => { jsonl += d; });
331
358
  child.stderr.on('data', (d) => { err += d; });
332
359
  child.on('error', (e) => {
333
360
  clearTimeout(killer);
@@ -340,7 +367,10 @@ function completeCodexCli({ system, prompt, model, timeoutMs }) {
340
367
  try { text = fs.readFileSync(outFile, 'utf8'); } catch (_e) { /* fall through */ }
341
368
  try { fs.unlinkSync(outFile); } catch (_e) { /* best effort */ }
342
369
  if (code !== 0) return reject(new Error(`codex exec exited ${code}: ${String(err).slice(0, 400)}`));
343
- resolve({ text: String(text).trim(), usage: null });
370
+ const ev = parseCodexJsonl(jsonl);
371
+ // Prefer the -o file; fall back to the stream's agent_message if it is empty.
372
+ const finalText = String(text || ev.text || '').trim();
373
+ resolve({ text: finalText, usage: ev.usage });
344
374
  });
345
375
  child.stdin.on('error', () => { /* ignore EPIPE if codex exits before reading all input */ });
346
376
  child.stdin.write(fullPrompt);
package/lib/receipt.js CHANGED
@@ -14,7 +14,8 @@ const SCHEMA_FILES = {
14
14
  '0.1': 'receipt.v0.1.schema.json',
15
15
  '0.2': 'receipt.v0.2.schema.json',
16
16
  '0.3': 'receipt.v0.3.schema.json',
17
- '0.3.1': 'receipt.schema.json',
17
+ '0.3.1': 'receipt.v0.3.1.schema.json',
18
+ '0.4': 'receipt.schema.json',
18
19
  };
19
20
 
20
21
  const _validators = {};
@@ -80,7 +81,7 @@ function aggregate(caseResults) {
80
81
  // run: { model_id, model_release_date, surface, runner_version, date_utc, judge, registry, transcripts }
81
82
  // cases: [ { id, mode, outcome, score, mean, stddev, samples, generation_hash, judge_sample_hashes, threshold, reason, judge } ]
82
83
  // editorialReviews: optional [ { url, source, date } ]
83
- function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED', editorialReviews = null }) {
84
+ function buildReceipt({ skill, suite, run, cases, economics = null, verificationLevel = 'TESTED', editorialReviews = null }) {
84
85
  // v0.3.1: cases marked failed_timeout are recorded in results.cases but EXCLUDED
85
86
  // from aggregates — a band is never fabricated from a case that did not complete.
86
87
  const okCases = cases.filter((c) => c.case_status !== 'failed_timeout');
@@ -138,6 +139,11 @@ function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED',
138
139
  // skill.tokens — estimated SKILL.md token size (value-per-token axis).
139
140
  if (run.surface_overhead_note) receipt.run.surface_overhead_note = run.surface_overhead_note;
140
141
  if (skill.tokens != null) receipt.skill.tokens = skill.tokens;
142
+ // v0.4 economics (additive-optional): the frozen prices this receipt's derived
143
+ // dollar figures were computed from, and the derived block itself. A receipt
144
+ // from a surface that reports no usage simply omits both.
145
+ if (run.pricing_snapshot) receipt.run.pricing_snapshot = run.pricing_snapshot;
146
+ if (economics) receipt.economics = economics;
141
147
  // v0.3.1: mark the receipt incomplete when any case failed (excluded above).
142
148
  if (failedCount > 0) {
143
149
  receipt.run.status = 'incomplete';
@@ -0,0 +1,162 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ const { bandVerdict, round } = require('./stats');
5
+ const { EFFECT_FLOOR } = require('../config');
6
+
7
+ // Revision drift — the fifth report type (spec 009, Report #006).
8
+ //
9
+ // Every other report type holds the skill text fixed and moves something
10
+ // underneath it: the model release (#001, #003), the vendor surface (#002), the
11
+ // capability tier (#004), the axes and the price (#005). This one inverts the
12
+ // design. The substrate is held still — same model, same provider, same surface,
13
+ // same suite, same fixed judge, same sampling — and the SKILL'S OWN TEXT moves,
14
+ // from the revision a report pinned to the revision upstream ships today.
15
+ //
16
+ // This module holds the report type's LANGUAGE as code rather than as
17
+ // hand-written page copy. That is deliberate: the fairness rule below is the one
18
+ // a measurement project is most tempted to apply in one direction only, and a
19
+ // rule that lives in prose cannot be gated before the run that would tempt it.
20
+
21
+ // ── the cell headline ────────────────────────────────────────────────────────
22
+ // A summary of the per-case band-overlap verdicts, worded about the REVISION.
23
+ // The release-drift headline says "the skill is measurably weaker", which is a
24
+ // sentence about a skill under a moving model. Here the model is the control.
25
+ function revisionHeadline(perCase) {
26
+ const reg = perCase.filter((r) => r.verdict === 'regression').length;
27
+ const imp = perCase.filter((r) => r.verdict === 'improvement').length;
28
+ const s = (n) => (n === 1 ? '' : 's');
29
+ if (reg && imp) {
30
+ return `MIXED — the revision improved ${imp} case${s(imp)} and regressed ${reg} on non-overlapping bands.`;
31
+ }
32
+ if (reg) {
33
+ return `REVISION REGRESSED — ${reg} case${s(reg)} scored lower under the current upstream text (bands do not overlap).`;
34
+ }
35
+ if (imp) {
36
+ return `REVISION IMPROVED — ${imp} case${s(imp)} scored higher under the current upstream text (bands do not overlap); none regressed.`;
37
+ }
38
+ return 'WITHIN NOISE — the revision moved no case beyond its confidence band; the pinned text and the current text measure the same.';
39
+ }
40
+
41
+ // Classification word for a cell, from its headline. Kept separate so a caller
42
+ // can branch on the class without parsing prose.
43
+ function revisionClass(perCase) {
44
+ const h = revisionHeadline(perCase);
45
+ return h.split(' —')[0];
46
+ }
47
+
48
+ // ── the fairness sentence ────────────────────────────────────────────────────
49
+ // Spec 009 § Fairness, clauses 1 and 4. Where a revision IMPROVED a skill, the
50
+ // published #005 figure understates the pack a reader can install today; where it
51
+ // REGRESSED one, #005 overstates it. Both sentences are generated by the same
52
+ // function, from the same template, so the disclosure cannot quietly become a
53
+ // one-directional courtesy — the symmetry is a property of the code, and the
54
+ // gate asserts it.
55
+ //
56
+ // A cell within noise gets NO sentence. #005's figure stands unamended, because
57
+ // nothing was measured that would amend it, and manufacturing a hedge for a null
58
+ // result is how a report launders noise into a finding.
59
+ function fairnessSentence({ slug, classification, report005Delta, measuredDelta }) {
60
+ const cls = String(classification || '');
61
+ if (cls !== 'REVISION IMPROVED' && cls !== 'REVISION REGRESSED') return null;
62
+ const improved = cls === 'REVISION IMPROVED';
63
+ const direction = improved ? 'understates' : 'overstates';
64
+ const d = (n) => (n == null ? 'n/a' : (n >= 0 ? '+' : '') + Number(n).toFixed(3));
65
+ return `Report #005 measured ${slug} at ${d(report005Delta)} on the text it had pinned. `
66
+ + `Report #006 measures the current upstream revision at ${d(measuredDelta)} on the same substrate and the same suite. `
67
+ + `#005's published figure therefore ${direction} the pack upstream ships today for this skill, `
68
+ + `and is amended by this report rather than corrected in place.`;
69
+ }
70
+
71
+ // ── per-cell scoping disclosures ─────────────────────────────────────────────
72
+ // Spec 009 AC-10. One cell in Report #006 measures something other than what its
73
+ // upstream author changed it to do, and the reader looking at that row is the
74
+ // reader who needs to be told.
75
+ const SCOPING_NOTES = {
76
+ 'git-workflow-and-versioning':
77
+ 'This revision changes the frontmatter `description:` line. In a skill runtime a description is a '
78
+ + 'routing trigger: it decides whether the skill loads, and never reaches the model as guidance. '
79
+ + 'Driftproof makes no routing decision — it always injects the skill, and passes the whole file, '
80
+ + 'frontmatter included, as the system prompt. This cell therefore measures the revision as added '
81
+ + 'context and cannot measure it as a trigger.',
82
+ };
83
+ function scopingNote(slug) {
84
+ return SCOPING_NOTES[slug] || null;
85
+ }
86
+
87
+ // ── the baseline-reproduction control ────────────────────────────────────────
88
+ // Spec 009 AC-6, and the thing that makes the free pinned arm honest.
89
+ //
90
+ // Report #006 reuses #005's receipts as the pinned-text arm. That is valid only
91
+ // if the substrate has not moved, and `run.model_release_date` is null on every
92
+ // #005 receipt, so id equality is the only version evidence a receipt carries. A
93
+ // provider that re-points a concrete id at a new snapshot is invisible to it.
94
+ //
95
+ // It does not have to be. Every fresh run emits a BASELINE arm: the same cases,
96
+ // the same substrate, and no skill text at all. The revision cannot touch it by
97
+ // construction, so comparing the fresh baseline against the reused receipt's
98
+ // baseline re-measures exactly the thing id equality could not prove — at no
99
+ // extra cost, because that arm is already paid for.
100
+ //
101
+ // A cell whose baselines do not reproduce is NOT MEASURED. The reuse is a tested
102
+ // prediction, not an assumption the report asks the reader to grant.
103
+ function baselineBands(receipt) {
104
+ const out = {};
105
+ for (const c of receipt.results.cases) {
106
+ if (c.mode !== 'baseline') continue;
107
+ out[c.id] = { mean: c.mean != null ? c.mean : c.score, stddev: c.stddev || 0 };
108
+ }
109
+ return out;
110
+ }
111
+
112
+ function baselineControl(reused, fresh) {
113
+ const A = baselineBands(reused);
114
+ const B = baselineBands(fresh);
115
+ const ids = [...new Set([...Object.keys(A), ...Object.keys(B)])];
116
+
117
+ const perCase = ids.map((id) => {
118
+ const before = A[id] || null;
119
+ const after = B[id] || null;
120
+ if (!before || !after) return { id, before, after, delta: null, moved: false, missing: true };
121
+ const delta = round(after.mean - before.mean);
122
+ // The same rule the study uses everywhere else: band separation AND the
123
+ // effect floor. A baseline that wobbles inside its band has not moved.
124
+ const raw = bandVerdict(before.mean, before.stddev, after.mean, after.stddev);
125
+ const separated = raw === 'regression' || raw === 'improvement';
126
+ return { id, before, after, delta, moved: separated && Math.abs(delta) >= EFFECT_FLOOR, missing: false };
127
+ });
128
+
129
+ const movedCases = perCase.filter((r) => r.moved);
130
+ const missing = perCase.filter((r) => r.missing);
131
+ const reproduced = movedCases.length === 0 && missing.length === 0;
132
+ const aggDelta = round(
133
+ (fresh.comparison && fresh.comparison.baseline_score != null ? fresh.comparison.baseline_score : 0)
134
+ - (reused.comparison && reused.comparison.baseline_score != null ? reused.comparison.baseline_score : 0),
135
+ );
136
+
137
+ return {
138
+ reproduced,
139
+ blocked: !reproduced,
140
+ verdict: reproduced ? 'MEASURED' : 'NOT MEASURED',
141
+ moved_cases: movedCases.map((r) => r.id),
142
+ missing_cases: missing.map((r) => r.id),
143
+ aggregate_baseline_delta: aggDelta,
144
+ floor: EFFECT_FLOOR,
145
+ // THE CONTROL PROVES NON-REPRODUCTION. IT CANNOT SAY WHY. These strings used
146
+ // to read 'the substrate moved' and 'the substrate held still' — a cause,
147
+ // asserted by a comparison that measures two baseline arms and nothing else.
148
+ // A 120-call stability probe then found generation-level sampling noise large
149
+ // enough to account for every gap this control saw, with no substrate movement
150
+ // required, and the report page retracted the claim while three committed
151
+ // control records still carried it (approval finding F-009-N). Reason strings
152
+ // only: no score, sample, hash or verdict changed with this edit.
153
+ reason: reproduced
154
+ ? 'the fresh baseline reproduces the reused receipt\'s baseline within the band and the floor, so the pinned-arm reuse stands for this cell'
155
+ : `the fresh baseline does not reproduce the reused receipt's baseline (${movedCases.length} case(s) moved beyond the band and the ${EFFECT_FLOOR} floor${missing.length ? `, ${missing.length} case(s) absent on one side` : ''}) — the reused pinned arm is not comparable to the fresh arm, so revision drift cannot be separated from whatever else changed in this cell; the control establishes non-reproduction and does not identify a cause`,
156
+ perCase,
157
+ };
158
+ }
159
+
160
+ module.exports = {
161
+ revisionHeadline, revisionClass, fairnessSentence, scopingNote, baselineControl,
162
+ };
package/lib/run.js CHANGED
@@ -1,14 +1,16 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
- const { complete, resolveModel, surfaceForModel, CODEX_OVERHEAD_NOTE } = require('./provider');
4
+ const { complete, resolveModel, surfaceForModel, isMeteredSurface, CODEX_OVERHEAD_NOTE } = require('./provider');
5
5
  const { gradeSamples, judgeSettings } = require('./judge');
6
6
  const { buildReceipt } = require('./receipt');
7
7
  const { sha256 } = require('./canonical');
8
- const { registryStatus, providerForModel } = require('./models');
8
+ const { registryStatus, providerForModel, priceForModel } = require('./models');
9
9
  const { perCallCostUSD } = require('./cost');
10
10
  const { runChecks } = require('./checks');
11
11
  const { estimateTokens } = require('./skillCost');
12
+ const { hasUsage, normalizeUsage } = require('./usage');
13
+ const { buildPricingSnapshot, computeEconomics } = require('./value');
12
14
  const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES } = require('../config');
13
15
 
14
16
  // Known model release dates (best-effort; null when unknown). Recorded into the
@@ -51,8 +53,8 @@ async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs }) {
51
53
  throw e;
52
54
  }
53
55
  const system = withSkill ? skillMd : undefined;
54
- const { text, usage, attempts } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
55
- return { text, usage, attempts };
56
+ const { text, usage, wall_ms, attempts } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
57
+ return { text, usage, wall_ms, attempts };
56
58
  }
57
59
 
58
60
  // Determine a case outcome from its sampled band and threshold.
@@ -91,6 +93,8 @@ async function judgeCase({ caseObj, response, generationHash, judgeModel, mode,
91
93
  // v0.3.1 deterministic post-checks (supplementary; NOT folded into `outcome`).
92
94
  const checks = runChecks(response, caseObj.checks);
93
95
  if (checks.length) caseResult.checks = checks;
96
+ // v0.4: grading overhead for this case row, kept OUT of the skill-value math.
97
+ if (hasUsage(g.usage)) caseResult.judge_usage = g.usage;
94
98
  return { caseResult, sampleTexts: g.sample_texts, attempts: g.attempts };
95
99
  }
96
100
 
@@ -170,6 +174,9 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
170
174
  const generationHash = sha256(String(gen.text || ''));
171
175
  onProgress({ case: c.id, mode, phase: 'judge', samples });
172
176
  const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ct, samples });
177
+ // v0.4: the GENERATION call's usage is the skill-value measurement (the
178
+ // judge's own usage rides separately on judge_usage, above).
179
+ if (hasUsage(gen.usage)) jr.caseResult.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
173
180
  calls += samples;
174
181
  // Live budget: count all judge calls (retries included) for this (case, mode).
175
182
  if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judgeModel, 'judge'));
@@ -198,6 +205,24 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
198
205
  const transcripts = keepTranscripts ? pairs.map((p) => p.transcript) : null;
199
206
 
200
207
  const surface = surfaceForModel(modelId);
208
+ const nowIso = opts.nowIso || new Date().toISOString();
209
+ // v0.4 economics. The pricing snapshot is frozen HERE, at run time, from the
210
+ // registry; every derived dollar figure below is computed from the snapshot and
211
+ // never from the live registry, so this receipt keeps its meaning when prices
212
+ // later change.
213
+ const pricingSnapshot = buildPricingSnapshot({
214
+ models: [modelId, judgeModel],
215
+ lookup: priceForModel,
216
+ nowIso,
217
+ });
218
+ const economics = computeEconomics({
219
+ cases: caseResults,
220
+ modelId,
221
+ judgeModelId: judgeModel,
222
+ pricingSnapshot,
223
+ surface,
224
+ meteredSurface: isMeteredSurface(surface),
225
+ });
201
226
  const receipt = buildReceipt({
202
227
  skill: {
203
228
  name: skill.name, version: skill.version, contentHash: skill.contentHash,
@@ -213,12 +238,14 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
213
238
  // v0.3.1: on the openai/cli (codex) surface, record the fixed harness preamble.
214
239
  surface_overhead_note: surface === 'openai-cli' ? CODEX_OVERHEAD_NOTE : undefined,
215
240
  runner_version: RUNNER_VERSION,
216
- date_utc: opts.nowIso || new Date().toISOString(),
241
+ date_utc: nowIso,
217
242
  registry: registryStatus(modelId),
218
243
  transcripts: keepTranscripts ? 'retained-local' : 'hashes-only',
219
244
  judge: judgeSettings(samples, judgeModel),
245
+ pricing_snapshot: pricingSnapshot,
220
246
  },
221
247
  cases: caseResults,
248
+ economics,
222
249
  verificationLevel: 'TESTED',
223
250
  });
224
251
 
package/lib/stub.js CHANGED
@@ -39,16 +39,37 @@ function stubComplete({ system, prompt }) {
39
39
  pass: score >= 0.7,
40
40
  reason: `stub judge (DRIFTPROOF_STUB): ${helped ? 'with-skill marker present' : 'baseline'}`,
41
41
  };
42
- return { text: JSON.stringify(body), usage: null };
42
+ return { text: JSON.stringify(body), usage: stubUsage({ kind: 'judge', prompt }) };
43
43
  }
44
44
  // Generation call: canned, marked by mode.
45
45
  const mark = system ? MARK_SKILL : MARK_BASE;
46
46
  const text = `${mark}\nfeat(stub): canned offline generation for CI (no model was called)`;
47
- return { text, usage: null };
47
+ return { text, usage: stubUsage({ kind: 'gen', system, prompt }) };
48
+ }
49
+
50
+ // v0.4: deterministic synthetic usage, so a zero-model-call stub run still
51
+ // exercises the whole economics path (usage → pricing snapshot → derived cost/
52
+ // latency fields → the value report's three axes). The numbers are SYNTHETIC and
53
+ // deliberately shaped like the real surfaces: a large fixed harness preamble that
54
+ // is identical in both arms, plus the with-skill arm's extra SKILL.md input and
55
+ // its slightly longer, slower output. No randomness — same input, same usage.
56
+ const STUB_HARNESS_PREAMBLE_TOKENS = 20000; // the fixed, arm-identical overhead
57
+ function stubUsage({ kind, system, prompt }) {
58
+ const promptTokens = Math.ceil(String(prompt || '').length / 4);
59
+ if (kind === 'judge') {
60
+ return { input_tokens: 1200 + promptTokens, output_tokens: 60, cached_tokens: 800, wall_ms: 900 };
61
+ }
62
+ const skillTokens = system ? Math.ceil(String(system).length / 4) : 0;
63
+ return {
64
+ input_tokens: STUB_HARNESS_PREAMBLE_TOKENS + promptTokens + skillTokens,
65
+ output_tokens: system ? 420 : 350,
66
+ cached_tokens: STUB_HARNESS_PREAMBLE_TOKENS,
67
+ wall_ms: system ? 4200 : 3600,
68
+ };
48
69
  }
49
70
 
50
71
  function stubEnabled() {
51
72
  return process.env.DRIFTPROOF_STUB === '1';
52
73
  }
53
74
 
54
- module.exports = { stubComplete, stubEnabled, MARK_SKILL, MARK_BASE };
75
+ module.exports = { stubComplete, stubEnabled, stubUsage, MARK_SKILL, MARK_BASE, STUB_HARNESS_PREAMBLE_TOKENS };
package/lib/usage.js ADDED
@@ -0,0 +1,168 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // Usage capture — normalizing what each surface reports about a single call.
5
+ //
6
+ // Every lane reports token usage in its own shape, and two of them (the CLI
7
+ // lanes) were previously discarding it entirely: the `claude -p` default text
8
+ // output carries no usage block at all, and `codex exec --json` streams usage on
9
+ // stdout that we were routing to /dev/null. Spec v0.4 captures it.
10
+ //
11
+ // ONE normalized shape, so three substrates are comparable:
12
+ //
13
+ // { input_tokens, output_tokens, cached_tokens, wall_ms }
14
+ //
15
+ // input_tokens TOTAL input presented to the model for this call, INCLUDING
16
+ // any portion served from cache. This is the field that most
17
+ // needs normalizing: the surfaces disagree about it (below).
18
+ // cached_tokens the portion of input_tokens served from cache (null when the
19
+ // surface does not surface it — never guessed, never 0-filled).
20
+ // output_tokens tokens generated, INCLUDING reasoning/thinking tokens where
21
+ // the surface bundles them (both CLI lanes do; the split is
22
+ // reported by the surface but deliberately not modelled here —
23
+ // see the disclosure in REPORT-STYLE.md).
24
+ // wall_ms measured BY US around the call (see lib/provider), not taken
25
+ // from the surface. It is the only field whose definition is
26
+ // identical across lanes, which is exactly why latency is
27
+ // reported as an observed, surface-disclosed number.
28
+ //
29
+ // THE INPUT-TOKEN DISAGREEMENT (verified against real output, 2026-08-18):
30
+ // - `claude -p --output-format json` reports input_tokens EXCLUDING cache:
31
+ // total input = input_tokens + cache_creation_input_tokens + cache_read_input_tokens.
32
+ // (Observed: 10 + 6,974 + 18,134 = 25,118 for a two-word prompt.)
33
+ // - `codex exec --json` reports input_tokens INCLUDING cache, with
34
+ // cached_input_tokens as a SUBSET of it. (Observed: 10,807 of which 4,480 cached.)
35
+ // Normalizing to "total including cache" makes the two comparable; the raw
36
+ // fixtures both parsers are tested against are checked in under tests/fixtures/.
37
+ //
38
+ // Both CLI surfaces prepend a large fixed harness preamble that we do not
39
+ // control (~25k tokens on claude-cli, ~11k on codex). It is IDENTICAL in the
40
+ // with-skill and baseline arms, so it cancels in the incremental (Δ) figures —
41
+ // which is why Δcost is the honest axis and absolute per-call cost is not.
42
+
43
+ function n(x) { return Number.isFinite(Number(x)) ? Number(x) : 0; }
44
+
45
+ // The empty/unknown usage record. Deliberately null (not zeros) so "the surface
46
+ // did not tell us" never reads as "the call cost nothing".
47
+ function emptyUsage() {
48
+ return { input_tokens: null, output_tokens: null, cached_tokens: null, wall_ms: null };
49
+ }
50
+
51
+ function hasUsage(u) {
52
+ return !!(u && (u.input_tokens != null || u.output_tokens != null));
53
+ }
54
+
55
+ // ── anthropic/cli — `claude -p --output-format json` ──────────────────────────
56
+ // The whole call is one JSON object; `result` carries the final text and `usage`
57
+ // the token counts. input_tokens EXCLUDES cache, so both cache fields are added
58
+ // back to reach the total actually presented to the model.
59
+ function parseClaudeCliJson(stdout) {
60
+ let j;
61
+ try { j = JSON.parse(String(stdout || '')); } catch (_e) { return null; }
62
+ if (!j || typeof j !== 'object') return null;
63
+ const u = j.usage || {};
64
+ const cacheRead = n(u.cache_read_input_tokens);
65
+ const cacheCreate = n(u.cache_creation_input_tokens);
66
+ const usage = {
67
+ input_tokens: n(u.input_tokens) + cacheRead + cacheCreate,
68
+ output_tokens: n(u.output_tokens),
69
+ // Cache READ is the portion genuinely served from cache. Cache CREATION was
70
+ // processed fresh this call (and billed at a premium), so it is not cached.
71
+ cached_tokens: cacheRead,
72
+ wall_ms: null,
73
+ };
74
+ return {
75
+ text: typeof j.result === 'string' ? j.result : '',
76
+ usage,
77
+ isError: j.is_error === true,
78
+ // The CLI reports its own dollar figure. Recorded here for completeness but
79
+ // NOT used: costs are computed uniformly from the frozen pricing snapshot so
80
+ // three substrates are on one basis (see lib/value.js).
81
+ surfaceReportedCostUsd: Number.isFinite(Number(j.total_cost_usd)) ? Number(j.total_cost_usd) : null,
82
+ };
83
+ }
84
+
85
+ // ── openai/cli — `codex exec --json` JSONL event stream ───────────────────────
86
+ // One JSON object per line. Usage rides the terminal `turn.completed` event;
87
+ // input_tokens already INCLUDES cached_input_tokens. Unparseable lines are
88
+ // skipped (the stream also carries progress events we do not model).
89
+ function parseCodexJsonl(stdout) {
90
+ const lines = String(stdout || '').split('\n').map((l) => l.trim()).filter(Boolean);
91
+ let usage = null;
92
+ let text = '';
93
+ for (const line of lines) {
94
+ let ev;
95
+ try { ev = JSON.parse(line); } catch (_e) { continue; }
96
+ if (!ev || typeof ev !== 'object') continue;
97
+ if (ev.type === 'turn.completed' && ev.usage) {
98
+ const u = ev.usage;
99
+ usage = {
100
+ input_tokens: n(u.input_tokens), // already total (cache included)
101
+ output_tokens: n(u.output_tokens), // includes reasoning_output_tokens
102
+ cached_tokens: u.cached_input_tokens == null ? null : n(u.cached_input_tokens),
103
+ wall_ms: null,
104
+ };
105
+ }
106
+ // The final message is normally read from the -o file; this is a fallback.
107
+ if (ev.type === 'item.completed' && ev.item && ev.item.type === 'agent_message' && typeof ev.item.text === 'string') {
108
+ text = ev.item.text;
109
+ }
110
+ }
111
+ return { usage, text };
112
+ }
113
+
114
+ // ── anthropic/api ─────────────────────────────────────────────────────────────
115
+ // The Messages API reports input_tokens EXCLUDING cache, same as the CLI.
116
+ function parseAnthropicApiUsage(u) {
117
+ if (!u) return emptyUsage();
118
+ const cacheRead = n(u.cache_read_input_tokens);
119
+ const cacheCreate = n(u.cache_creation_input_tokens);
120
+ return {
121
+ input_tokens: n(u.input_tokens) + cacheRead + cacheCreate,
122
+ output_tokens: n(u.output_tokens),
123
+ cached_tokens: u.cache_read_input_tokens == null ? null : cacheRead,
124
+ wall_ms: null,
125
+ };
126
+ }
127
+
128
+ // ── openai/api (Chat Completions-compatible) ──────────────────────────────────
129
+ // prompt_tokens is the total; the cached portion, when present, is nested under
130
+ // prompt_tokens_details.cached_tokens.
131
+ function parseOpenaiApiUsage(u) {
132
+ if (!u) return emptyUsage();
133
+ const details = u.prompt_tokens_details || {};
134
+ return {
135
+ input_tokens: n(u.prompt_tokens),
136
+ output_tokens: n(u.completion_tokens),
137
+ cached_tokens: details.cached_tokens == null ? null : n(details.cached_tokens),
138
+ wall_ms: null,
139
+ };
140
+ }
141
+
142
+ // Sum a list of usage records into one (used for the N judge samples of a case).
143
+ // Null-safe: unknown fields stay null unless at least one record reported them.
144
+ function sumUsage(list) {
145
+ const records = (list || []).filter(Boolean);
146
+ if (!records.length) return emptyUsage();
147
+ const out = { input_tokens: null, output_tokens: null, cached_tokens: null, wall_ms: null };
148
+ for (const key of ['input_tokens', 'output_tokens', 'cached_tokens', 'wall_ms']) {
149
+ const present = records.filter((r) => r[key] != null);
150
+ if (present.length) out[key] = present.reduce((a, r) => a + n(r[key]), 0);
151
+ }
152
+ return out;
153
+ }
154
+
155
+ // Round a usage record's fields to integers (tokens and ms are whole units).
156
+ function normalizeUsage(u) {
157
+ if (!u) return emptyUsage();
158
+ const out = emptyUsage();
159
+ for (const key of ['input_tokens', 'output_tokens', 'cached_tokens', 'wall_ms']) {
160
+ if (u[key] != null && Number.isFinite(Number(u[key]))) out[key] = Math.round(Number(u[key]));
161
+ }
162
+ return out;
163
+ }
164
+
165
+ module.exports = {
166
+ emptyUsage, hasUsage, sumUsage, normalizeUsage,
167
+ parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage,
168
+ };