driftproof 0.4.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -3
- package/config.js +18 -2
- package/lib/judge.js +11 -3
- package/lib/provider.js +55 -25
- package/lib/receipt.js +8 -2
- package/lib/run.js +32 -5
- package/lib/stub.js +24 -3
- package/lib/usage.js +168 -0
- package/lib/value.js +502 -0
- package/package.json +1 -1
- package/spec/RECEIPT.md +50 -7
- package/spec/receipt.schema.json +703 -76
- package/spec/receipt.v0.3.1.schema.json +642 -0
package/README.md
CHANGED
|
@@ -162,7 +162,7 @@ A receipt is the unit of evidence — one JSON document conforming to
|
|
|
162
162
|
|
|
163
163
|
```jsonc
|
|
164
164
|
{
|
|
165
|
-
"schema_version": "0.
|
|
165
|
+
"schema_version": "0.4",
|
|
166
166
|
"skill": { "name": "commit-message-conventions", "version": "0.2.0",
|
|
167
167
|
"content_hash": "…sha256 over SKILL.md + bundled files…" },
|
|
168
168
|
"suite": { "format": "agentskills.io/evals", "suite_hash": "…", "case_count": 10 },
|
|
@@ -171,7 +171,7 @@ A receipt is the unit of evidence — one JSON document conforming to
|
|
|
171
171
|
"model_release_date": "2025-10-01",
|
|
172
172
|
"provider": "anthropic",
|
|
173
173
|
"surface": "claude-cli",
|
|
174
|
-
"runner_version": "0.
|
|
174
|
+
"runner_version": "0.5.0",
|
|
175
175
|
"date_utc": "2026-07-27T…Z",
|
|
176
176
|
"registry": "registered",
|
|
177
177
|
"transcripts": "hashes-only",
|
|
@@ -233,7 +233,7 @@ jobs:
|
|
|
233
233
|
runs-on: ubuntu-latest
|
|
234
234
|
steps:
|
|
235
235
|
- uses: actions/checkout@v4
|
|
236
|
-
- uses: driftproofhq/driftproof@v0.
|
|
236
|
+
- uses: driftproofhq/driftproof@v0.5.0
|
|
237
237
|
with:
|
|
238
238
|
skill-dir: skills/my-skill
|
|
239
239
|
models: claude-haiku-4-5
|
package/config.js
CHANGED
|
@@ -9,7 +9,7 @@ const PROJECT_NAME = 'driftproof';
|
|
|
9
9
|
// Bumped whenever the runner's behaviour or receipt-generation semantics change
|
|
10
10
|
// in a way that could affect results. Recorded into every receipt as
|
|
11
11
|
// run.runner_version so a receipt is reproducible against a known engine.
|
|
12
|
-
const RUNNER_VERSION = '0.
|
|
12
|
+
const RUNNER_VERSION = '0.5.0';
|
|
13
13
|
|
|
14
14
|
// The eval format we CONSUME (we deliberately do not invent our own).
|
|
15
15
|
const SUITE_FORMAT = 'agentskills.io/evals';
|
|
@@ -22,7 +22,14 @@ const SUITE_FORMAT = 'agentskills.io/evals';
|
|
|
22
22
|
// surface enums (openai-api/openai-cli), optional run.surface_overhead_note,
|
|
23
23
|
// optional per-case checks[] (deterministic post-checks), and optional
|
|
24
24
|
// skill.tokens (value-per-token axis). v0.1/v0.2/v0.3 receipts still load.
|
|
25
|
-
|
|
25
|
+
// v0.4 (additive over v0.3.1) adds the ECONOMICS axis: per-case, per-arm
|
|
26
|
+
// generation `usage` (input/output/cached tokens + measured wall_ms), a
|
|
27
|
+
// separate per-case `judge_usage` (measurement overhead, excluded from
|
|
28
|
+
// every skill-value figure), run.pricing_snapshot (registry prices frozen
|
|
29
|
+
// at run time so derived dollars stay reproducible), and the derived
|
|
30
|
+
// `economics` block. v0.3.1 is frozen as receipt.v0.3.1.schema.json;
|
|
31
|
+
// v0.1/v0.2/v0.3/v0.3.1 receipts all still load.
|
|
32
|
+
const RECEIPT_SCHEMA_VERSION = '0.4';
|
|
26
33
|
|
|
27
34
|
// Hard USD budget defaults per entry point (Week 4). --max-usd overrides any of
|
|
28
35
|
// these. The projection is refused before any call if it exceeds the cap, on
|
|
@@ -73,10 +80,19 @@ const REPORT_004_BASE_MODEL = 'claude-opus-5'; // flagship tier
|
|
|
73
80
|
const REPORT_004_FRONTIER_MODEL = 'claude-fable-5'; // frontier tier (full id — no alias)
|
|
74
81
|
const REPORT_004_JUDGE_MODEL = 'claude-haiku-4-5';
|
|
75
82
|
|
|
83
|
+
// Report #005 is a VALUE report — the fourth report type. It asks what a skill
|
|
84
|
+
// COSTS to run alongside whether it helps, over three substrates, and shows the
|
|
85
|
+
// three axes (accuracy lift / Δcost / Δlatency) side by side and never combined.
|
|
86
|
+
// Same suites, same fixed judge; the substrate list spans two providers so the
|
|
87
|
+
// economics are read across surfaces, not within one vendor's pricing.
|
|
88
|
+
const REPORT_005_MODELS = ['claude-sonnet-5', 'claude-fable-5', 'gpt-5.6-sol'];
|
|
89
|
+
const REPORT_005_JUDGE_MODEL = 'claude-haiku-4-5';
|
|
90
|
+
|
|
76
91
|
module.exports = {
|
|
77
92
|
PROJECT_NAME, RUNNER_VERSION, SUITE_FORMAT, RECEIPT_SCHEMA_VERSION, DEFAULT_JUDGE_SAMPLES,
|
|
78
93
|
EFFECT_FLOOR, DEV_MAX_USD, REPORT_MAX_USD, TRIGGER_MAX_USD,
|
|
79
94
|
REPORT_002_CLAUDE_MODEL, REPORT_002_GPT_MODEL, REPORT_002_JUDGE_MODEL,
|
|
80
95
|
REPORT_003_NEW_MODEL, REPORT_003_OLD_MODEL, REPORT_003_JUDGE_MODEL,
|
|
81
96
|
REPORT_004_BASE_MODEL, REPORT_004_FRONTIER_MODEL, REPORT_004_JUDGE_MODEL,
|
|
97
|
+
REPORT_005_MODELS, REPORT_005_JUDGE_MODEL,
|
|
82
98
|
};
|
package/lib/judge.js
CHANGED
|
@@ -5,6 +5,7 @@ const { complete, surfaceForModel } = require('./provider');
|
|
|
5
5
|
const { extractJsonObject } = require('./json');
|
|
6
6
|
const { sha256 } = require('./canonical');
|
|
7
7
|
const { mean, stddev } = require('./stats');
|
|
8
|
+
const { sumUsage } = require('./usage');
|
|
8
9
|
|
|
9
10
|
// Rubric-based LLM judge.
|
|
10
11
|
//
|
|
@@ -82,16 +83,16 @@ function judgeSettings(samples, judgeModel) {
|
|
|
82
83
|
// transcript auditability, and optionally retained under --keep-transcripts).
|
|
83
84
|
async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature }) {
|
|
84
85
|
const prompt = buildJudgePrompt({ task, response, rubric });
|
|
85
|
-
const { text, attempts } = await complete({ system: JUDGE_SYSTEM, prompt, model, maxTokens: 400, timeoutMs, temperature });
|
|
86
|
+
const { text, attempts, usage } = await complete({ system: JUDGE_SYSTEM, prompt, model, maxTokens: 400, timeoutMs, temperature });
|
|
86
87
|
let parsed;
|
|
87
88
|
try {
|
|
88
89
|
parsed = extractJsonObject(text);
|
|
89
90
|
} catch (_e) {
|
|
90
91
|
// Unsalvageable judge output → conservative 0 (a judge that can't be parsed
|
|
91
92
|
// must never silently "pass"), tagged so the caller can see it happened.
|
|
92
|
-
return { score: 0, reason: 'judge output unparseable', unparsed: true, raw: String(text || ''), attempts: attempts || 1 };
|
|
93
|
+
return { score: 0, reason: 'judge output unparseable', unparsed: true, raw: String(text || ''), attempts: attempts || 1, usage };
|
|
93
94
|
}
|
|
94
|
-
return { score: clamp01(parsed.score), reason: String(parsed.reason || '').slice(0, 300), raw: String(text || ''), attempts: attempts || 1 };
|
|
95
|
+
return { score: clamp01(parsed.score), reason: String(parsed.reason || '').slice(0, 300), raw: String(text || ''), attempts: attempts || 1, usage };
|
|
95
96
|
}
|
|
96
97
|
|
|
97
98
|
// Grade a response N times and return the sampled distribution:
|
|
@@ -103,6 +104,7 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
|
|
|
103
104
|
const scores = [];
|
|
104
105
|
const reasons = [];
|
|
105
106
|
const rawTexts = [];
|
|
107
|
+
const usages = [];
|
|
106
108
|
let attemptsTotal = 0;
|
|
107
109
|
for (let i = 0; i < samples; i++) {
|
|
108
110
|
let r;
|
|
@@ -116,6 +118,7 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
|
|
|
116
118
|
throw e;
|
|
117
119
|
}
|
|
118
120
|
attemptsTotal += r.attempts || 1;
|
|
121
|
+
usages.push(r.usage || null);
|
|
119
122
|
scores.push(r.score);
|
|
120
123
|
reasons.push(r.reason);
|
|
121
124
|
rawTexts.push(r.raw || '');
|
|
@@ -134,6 +137,11 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
|
|
|
134
137
|
model_id: model,
|
|
135
138
|
rubric_hash: rubricHash(rubric),
|
|
136
139
|
attempts: attemptsTotal,
|
|
140
|
+
// v0.4: the measurement overhead of grading this one case — the SUM over all
|
|
141
|
+
// N judge calls. Recorded in the receipt as the case's `judge_usage` and
|
|
142
|
+
// EXCLUDED from every skill-value figure (lib/value.js): it is a cost we
|
|
143
|
+
// impose to measure, not a cost of running the skill.
|
|
144
|
+
usage: sumUsage(usages),
|
|
137
145
|
};
|
|
138
146
|
}
|
|
139
147
|
|
package/lib/provider.js
CHANGED
|
@@ -7,6 +7,9 @@ const path = require('path');
|
|
|
7
7
|
const { spawn } = require('child_process');
|
|
8
8
|
const { withRetry, withTimeout } = require('./json');
|
|
9
9
|
const { stubComplete, stubEnabled } = require('./stub');
|
|
10
|
+
const {
|
|
11
|
+
parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage, normalizeUsage,
|
|
12
|
+
} = require('./usage');
|
|
10
13
|
|
|
11
14
|
// Provider abstraction: one `complete()` call over a TWO-AXIS lane model —
|
|
12
15
|
// provider (anthropic | openai) × surface (api | cli). Four concrete lanes:
|
|
@@ -109,9 +112,18 @@ const CODEX_OVERHEAD_NOTE =
|
|
|
109
112
|
+ 'control; the model id is set by us via -m (it is not echoed in the JSONL stream). '
|
|
110
113
|
+ 'Approval prompting is off by default on `codex exec` (no -a flag is passed).';
|
|
111
114
|
|
|
112
|
-
// Send a single-turn prompt and return { text, usage, surface, provider }.
|
|
113
|
-
//
|
|
114
|
-
//
|
|
115
|
+
// Send a single-turn prompt and return { text, usage, wall_ms, surface, provider }.
|
|
116
|
+
//
|
|
117
|
+
// usage is the normalized v0.4 record { input_tokens, output_tokens, cached_tokens,
|
|
118
|
+
// wall_ms } on EVERY lane — including the two CLI lanes, which report it in their
|
|
119
|
+
// structured output (`claude -p --output-format json`; the `codex exec --json`
|
|
120
|
+
// JSONL stream). See lib/usage.js for the per-surface shapes and the input-token
|
|
121
|
+
// normalization. Fields the surface does not report stay null, never 0.
|
|
122
|
+
//
|
|
123
|
+
// wall_ms is measured HERE, around the successful attempt, so it means the same
|
|
124
|
+
// thing on all four lanes (retries are excluded — a retried call's latency would
|
|
125
|
+
// describe our backoff, not the model).
|
|
126
|
+
//
|
|
115
127
|
// `temperature` is honoured ONLY on api surfaces; cli surfaces control sampling
|
|
116
128
|
// themselves, so temperature is ignored there and the receipt records that fact.
|
|
117
129
|
async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = null, temperature = undefined }) {
|
|
@@ -120,7 +132,10 @@ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = n
|
|
|
120
132
|
// Offline stub surface: canned completion, zero model calls. The receipt still
|
|
121
133
|
// records the real surface/provider so a stub run is not mistaken for a genuine
|
|
122
134
|
// one at read time — only the generation/judge TEXT is canned.
|
|
123
|
-
if (stubEnabled())
|
|
135
|
+
if (stubEnabled()) {
|
|
136
|
+
const s = stubComplete({ system, prompt });
|
|
137
|
+
return { ...s, usage: normalizeUsage(s.usage), wall_ms: (s.usage && s.usage.wall_ms) || 0, surface, provider, attempts: 1 };
|
|
138
|
+
}
|
|
124
139
|
|
|
125
140
|
// Per-surface tolerance (see retryPolicyForSurface): a cli/subscription surface
|
|
126
141
|
// spawns a first-party CLI subprocess whose cold-start is slow and occasionally
|
|
@@ -144,10 +159,20 @@ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = n
|
|
|
144
159
|
// throttling/cold-starts so one blip doesn't abort a multi-hour grind. `attempts`
|
|
145
160
|
// counts every try (retries included) so the budget can charge for them.
|
|
146
161
|
let attempts = 0;
|
|
147
|
-
|
|
162
|
+
// Wall-clock of the attempt that SUCCEEDED (each attempt overwrites, so a
|
|
163
|
+
// retried call reports the latency of the call that actually produced the text,
|
|
164
|
+
// not the accumulated backoff).
|
|
165
|
+
let wallMs = null;
|
|
166
|
+
const runner = () => {
|
|
167
|
+
attempts += 1;
|
|
168
|
+
const t0 = Date.now();
|
|
169
|
+
return withTimeout(laneRunner, effTimeout, `provider(${surface})`)
|
|
170
|
+
.then((r) => { wallMs = Date.now() - t0; return r; });
|
|
171
|
+
};
|
|
148
172
|
try {
|
|
149
173
|
const out = await withRetry(runner, { tries: policy.tries, baseDelayMs });
|
|
150
|
-
|
|
174
|
+
const usage = normalizeUsage({ ...(out.usage || {}), wall_ms: wallMs });
|
|
175
|
+
return { ...out, usage, wall_ms: wallMs, surface, provider, attempts };
|
|
151
176
|
} catch (e) {
|
|
152
177
|
// Surface the attempt count so a persistently-failing call can be charged for
|
|
153
178
|
// (and, for a timeout, marked failed_timeout by the runner instead of fatal).
|
|
@@ -172,13 +197,7 @@ async function completeAnthropicApi({ system, prompt, model, maxTokens, temperat
|
|
|
172
197
|
if (temperature !== undefined) params.temperature = temperature;
|
|
173
198
|
const resp = await client.messages.create(params);
|
|
174
199
|
const text = (resp.content || []).filter((b) => b.type === 'text').map((b) => b.text).join('');
|
|
175
|
-
return {
|
|
176
|
-
text,
|
|
177
|
-
usage: {
|
|
178
|
-
input_tokens: (resp.usage && resp.usage.input_tokens) || 0,
|
|
179
|
-
output_tokens: (resp.usage && resp.usage.output_tokens) || 0,
|
|
180
|
-
},
|
|
181
|
-
};
|
|
200
|
+
return { text, usage: parseAnthropicApiUsage(resp.usage) };
|
|
182
201
|
}
|
|
183
202
|
|
|
184
203
|
// ── anthropic/cli (claude -p) ──────────────────────────────────────────────────
|
|
@@ -189,7 +208,13 @@ function completeClaudeCli({ system, prompt, model, timeoutMs }) {
|
|
|
189
208
|
const env = { ...process.env };
|
|
190
209
|
delete env.ANTHROPIC_API_KEY;
|
|
191
210
|
|
|
192
|
-
|
|
211
|
+
// v0.4: `--output-format json` returns ONE JSON object carrying both the
|
|
212
|
+
// final text (`result`) and the token usage (`usage`) — the default text
|
|
213
|
+
// output carries no usage at all, which is why usage was previously null on
|
|
214
|
+
// this lane. The text is read from the parsed object; if the CLI ever emits
|
|
215
|
+
// something unparseable we fall back to the raw stdout so a run degrades to
|
|
216
|
+
// the old behaviour (text, no usage) rather than failing.
|
|
217
|
+
const args = ['-p', '--output-format', 'json', '--model', resolveModel(model)];
|
|
193
218
|
if (system) args.push('--append-system-prompt', system);
|
|
194
219
|
|
|
195
220
|
const child = spawn('claude', args, { env, stdio: ['pipe', 'pipe', 'pipe'] });
|
|
@@ -206,7 +231,10 @@ function completeClaudeCli({ system, prompt, model, timeoutMs }) {
|
|
|
206
231
|
child.on('close', (code) => {
|
|
207
232
|
clearTimeout(killer);
|
|
208
233
|
if (code !== 0) return reject(new Error(`claude CLI exited ${code}: ${err.slice(0, 400)}`));
|
|
209
|
-
|
|
234
|
+
const parsed = parseClaudeCliJson(out);
|
|
235
|
+
if (!parsed) return resolve({ text: out.trim(), usage: null });
|
|
236
|
+
if (parsed.isError) return reject(new Error(`claude CLI reported is_error: ${String(parsed.text || '').slice(0, 300)}`));
|
|
237
|
+
resolve({ text: String(parsed.text || '').trim(), usage: parsed.usage });
|
|
210
238
|
});
|
|
211
239
|
child.stdin.write(prompt);
|
|
212
240
|
child.stdin.end();
|
|
@@ -250,14 +278,7 @@ async function completeOpenaiApi({ system, prompt, model, maxTokens, temperature
|
|
|
250
278
|
let parsed;
|
|
251
279
|
try { parsed = JSON.parse(raw); } catch (_e) { throw new Error(`openai/api: unparseable response: ${raw.slice(0, 200)}`); }
|
|
252
280
|
const text = ((parsed.choices && parsed.choices[0] && parsed.choices[0].message && parsed.choices[0].message.content) || '');
|
|
253
|
-
|
|
254
|
-
return {
|
|
255
|
-
text: String(text).trim(),
|
|
256
|
-
usage: {
|
|
257
|
-
input_tokens: usage.prompt_tokens || 0,
|
|
258
|
-
output_tokens: usage.completion_tokens || 0,
|
|
259
|
-
},
|
|
260
|
-
};
|
|
281
|
+
return { text: String(text).trim(), usage: parseOpenaiApiUsage(parsed.usage) };
|
|
261
282
|
}
|
|
262
283
|
|
|
263
284
|
// Read the OpenAI provider config (base_url + api-key env) from the registry,
|
|
@@ -325,9 +346,15 @@ function completeCodexCli({ system, prompt, model, timeoutMs }) {
|
|
|
325
346
|
// stdin (per its --help), which is content-agnostic.
|
|
326
347
|
const args = codexFinalArgs({ model, outFile });
|
|
327
348
|
|
|
328
|
-
|
|
349
|
+
// v0.4: stdout is CAPTURED (it was 'ignore'). `--json` streams JSONL events
|
|
350
|
+
// there, and the terminal `turn.completed` event carries this call's token
|
|
351
|
+
// usage — the only place codex reports it. The final message still comes from
|
|
352
|
+
// the -o file (cleaner than scraping the stream); the JSONL is read for usage.
|
|
353
|
+
const child = spawn('codex', args, { env: { ...process.env }, stdio: ['pipe', 'pipe', 'pipe'] });
|
|
329
354
|
let err = '';
|
|
355
|
+
let jsonl = '';
|
|
330
356
|
const killer = setTimeout(() => { child.kill('SIGKILL'); }, timeoutMs + 5000);
|
|
357
|
+
child.stdout.on('data', (d) => { jsonl += d; });
|
|
331
358
|
child.stderr.on('data', (d) => { err += d; });
|
|
332
359
|
child.on('error', (e) => {
|
|
333
360
|
clearTimeout(killer);
|
|
@@ -340,7 +367,10 @@ function completeCodexCli({ system, prompt, model, timeoutMs }) {
|
|
|
340
367
|
try { text = fs.readFileSync(outFile, 'utf8'); } catch (_e) { /* fall through */ }
|
|
341
368
|
try { fs.unlinkSync(outFile); } catch (_e) { /* best effort */ }
|
|
342
369
|
if (code !== 0) return reject(new Error(`codex exec exited ${code}: ${String(err).slice(0, 400)}`));
|
|
343
|
-
|
|
370
|
+
const ev = parseCodexJsonl(jsonl);
|
|
371
|
+
// Prefer the -o file; fall back to the stream's agent_message if it is empty.
|
|
372
|
+
const finalText = String(text || ev.text || '').trim();
|
|
373
|
+
resolve({ text: finalText, usage: ev.usage });
|
|
344
374
|
});
|
|
345
375
|
child.stdin.on('error', () => { /* ignore EPIPE if codex exits before reading all input */ });
|
|
346
376
|
child.stdin.write(fullPrompt);
|
package/lib/receipt.js
CHANGED
|
@@ -14,7 +14,8 @@ const SCHEMA_FILES = {
|
|
|
14
14
|
'0.1': 'receipt.v0.1.schema.json',
|
|
15
15
|
'0.2': 'receipt.v0.2.schema.json',
|
|
16
16
|
'0.3': 'receipt.v0.3.schema.json',
|
|
17
|
-
'0.3.1': 'receipt.schema.json',
|
|
17
|
+
'0.3.1': 'receipt.v0.3.1.schema.json',
|
|
18
|
+
'0.4': 'receipt.schema.json',
|
|
18
19
|
};
|
|
19
20
|
|
|
20
21
|
const _validators = {};
|
|
@@ -80,7 +81,7 @@ function aggregate(caseResults) {
|
|
|
80
81
|
// run: { model_id, model_release_date, surface, runner_version, date_utc, judge, registry, transcripts }
|
|
81
82
|
// cases: [ { id, mode, outcome, score, mean, stddev, samples, generation_hash, judge_sample_hashes, threshold, reason, judge } ]
|
|
82
83
|
// editorialReviews: optional [ { url, source, date } ]
|
|
83
|
-
function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED', editorialReviews = null }) {
|
|
84
|
+
function buildReceipt({ skill, suite, run, cases, economics = null, verificationLevel = 'TESTED', editorialReviews = null }) {
|
|
84
85
|
// v0.3.1: cases marked failed_timeout are recorded in results.cases but EXCLUDED
|
|
85
86
|
// from aggregates — a band is never fabricated from a case that did not complete.
|
|
86
87
|
const okCases = cases.filter((c) => c.case_status !== 'failed_timeout');
|
|
@@ -138,6 +139,11 @@ function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED',
|
|
|
138
139
|
// skill.tokens — estimated SKILL.md token size (value-per-token axis).
|
|
139
140
|
if (run.surface_overhead_note) receipt.run.surface_overhead_note = run.surface_overhead_note;
|
|
140
141
|
if (skill.tokens != null) receipt.skill.tokens = skill.tokens;
|
|
142
|
+
// v0.4 economics (additive-optional): the frozen prices this receipt's derived
|
|
143
|
+
// dollar figures were computed from, and the derived block itself. A receipt
|
|
144
|
+
// from a surface that reports no usage simply omits both.
|
|
145
|
+
if (run.pricing_snapshot) receipt.run.pricing_snapshot = run.pricing_snapshot;
|
|
146
|
+
if (economics) receipt.economics = economics;
|
|
141
147
|
// v0.3.1: mark the receipt incomplete when any case failed (excluded above).
|
|
142
148
|
if (failedCount > 0) {
|
|
143
149
|
receipt.run.status = 'incomplete';
|
package/lib/run.js
CHANGED
|
@@ -1,14 +1,16 @@
|
|
|
1
1
|
// SPDX-License-Identifier: Apache-2.0
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
-
const { complete, resolveModel, surfaceForModel, CODEX_OVERHEAD_NOTE } = require('./provider');
|
|
4
|
+
const { complete, resolveModel, surfaceForModel, isMeteredSurface, CODEX_OVERHEAD_NOTE } = require('./provider');
|
|
5
5
|
const { gradeSamples, judgeSettings } = require('./judge');
|
|
6
6
|
const { buildReceipt } = require('./receipt');
|
|
7
7
|
const { sha256 } = require('./canonical');
|
|
8
|
-
const { registryStatus, providerForModel } = require('./models');
|
|
8
|
+
const { registryStatus, providerForModel, priceForModel } = require('./models');
|
|
9
9
|
const { perCallCostUSD } = require('./cost');
|
|
10
10
|
const { runChecks } = require('./checks');
|
|
11
11
|
const { estimateTokens } = require('./skillCost');
|
|
12
|
+
const { hasUsage, normalizeUsage } = require('./usage');
|
|
13
|
+
const { buildPricingSnapshot, computeEconomics } = require('./value');
|
|
12
14
|
const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES } = require('../config');
|
|
13
15
|
|
|
14
16
|
// Known model release dates (best-effort; null when unknown). Recorded into the
|
|
@@ -51,8 +53,8 @@ async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs }) {
|
|
|
51
53
|
throw e;
|
|
52
54
|
}
|
|
53
55
|
const system = withSkill ? skillMd : undefined;
|
|
54
|
-
const { text, usage, attempts } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
|
|
55
|
-
return { text, usage, attempts };
|
|
56
|
+
const { text, usage, wall_ms, attempts } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
|
|
57
|
+
return { text, usage, wall_ms, attempts };
|
|
56
58
|
}
|
|
57
59
|
|
|
58
60
|
// Determine a case outcome from its sampled band and threshold.
|
|
@@ -91,6 +93,8 @@ async function judgeCase({ caseObj, response, generationHash, judgeModel, mode,
|
|
|
91
93
|
// v0.3.1 deterministic post-checks (supplementary; NOT folded into `outcome`).
|
|
92
94
|
const checks = runChecks(response, caseObj.checks);
|
|
93
95
|
if (checks.length) caseResult.checks = checks;
|
|
96
|
+
// v0.4: grading overhead for this case row, kept OUT of the skill-value math.
|
|
97
|
+
if (hasUsage(g.usage)) caseResult.judge_usage = g.usage;
|
|
94
98
|
return { caseResult, sampleTexts: g.sample_texts, attempts: g.attempts };
|
|
95
99
|
}
|
|
96
100
|
|
|
@@ -170,6 +174,9 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
170
174
|
const generationHash = sha256(String(gen.text || ''));
|
|
171
175
|
onProgress({ case: c.id, mode, phase: 'judge', samples });
|
|
172
176
|
const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ct, samples });
|
|
177
|
+
// v0.4: the GENERATION call's usage is the skill-value measurement (the
|
|
178
|
+
// judge's own usage rides separately on judge_usage, above).
|
|
179
|
+
if (hasUsage(gen.usage)) jr.caseResult.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
|
|
173
180
|
calls += samples;
|
|
174
181
|
// Live budget: count all judge calls (retries included) for this (case, mode).
|
|
175
182
|
if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judgeModel, 'judge'));
|
|
@@ -198,6 +205,24 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
198
205
|
const transcripts = keepTranscripts ? pairs.map((p) => p.transcript) : null;
|
|
199
206
|
|
|
200
207
|
const surface = surfaceForModel(modelId);
|
|
208
|
+
const nowIso = opts.nowIso || new Date().toISOString();
|
|
209
|
+
// v0.4 economics. The pricing snapshot is frozen HERE, at run time, from the
|
|
210
|
+
// registry; every derived dollar figure below is computed from the snapshot and
|
|
211
|
+
// never from the live registry, so this receipt keeps its meaning when prices
|
|
212
|
+
// later change.
|
|
213
|
+
const pricingSnapshot = buildPricingSnapshot({
|
|
214
|
+
models: [modelId, judgeModel],
|
|
215
|
+
lookup: priceForModel,
|
|
216
|
+
nowIso,
|
|
217
|
+
});
|
|
218
|
+
const economics = computeEconomics({
|
|
219
|
+
cases: caseResults,
|
|
220
|
+
modelId,
|
|
221
|
+
judgeModelId: judgeModel,
|
|
222
|
+
pricingSnapshot,
|
|
223
|
+
surface,
|
|
224
|
+
meteredSurface: isMeteredSurface(surface),
|
|
225
|
+
});
|
|
201
226
|
const receipt = buildReceipt({
|
|
202
227
|
skill: {
|
|
203
228
|
name: skill.name, version: skill.version, contentHash: skill.contentHash,
|
|
@@ -213,12 +238,14 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
213
238
|
// v0.3.1: on the openai/cli (codex) surface, record the fixed harness preamble.
|
|
214
239
|
surface_overhead_note: surface === 'openai-cli' ? CODEX_OVERHEAD_NOTE : undefined,
|
|
215
240
|
runner_version: RUNNER_VERSION,
|
|
216
|
-
date_utc:
|
|
241
|
+
date_utc: nowIso,
|
|
217
242
|
registry: registryStatus(modelId),
|
|
218
243
|
transcripts: keepTranscripts ? 'retained-local' : 'hashes-only',
|
|
219
244
|
judge: judgeSettings(samples, judgeModel),
|
|
245
|
+
pricing_snapshot: pricingSnapshot,
|
|
220
246
|
},
|
|
221
247
|
cases: caseResults,
|
|
248
|
+
economics,
|
|
222
249
|
verificationLevel: 'TESTED',
|
|
223
250
|
});
|
|
224
251
|
|
package/lib/stub.js
CHANGED
|
@@ -39,16 +39,37 @@ function stubComplete({ system, prompt }) {
|
|
|
39
39
|
pass: score >= 0.7,
|
|
40
40
|
reason: `stub judge (DRIFTPROOF_STUB): ${helped ? 'with-skill marker present' : 'baseline'}`,
|
|
41
41
|
};
|
|
42
|
-
return { text: JSON.stringify(body), usage:
|
|
42
|
+
return { text: JSON.stringify(body), usage: stubUsage({ kind: 'judge', prompt }) };
|
|
43
43
|
}
|
|
44
44
|
// Generation call: canned, marked by mode.
|
|
45
45
|
const mark = system ? MARK_SKILL : MARK_BASE;
|
|
46
46
|
const text = `${mark}\nfeat(stub): canned offline generation for CI (no model was called)`;
|
|
47
|
-
return { text, usage:
|
|
47
|
+
return { text, usage: stubUsage({ kind: 'gen', system, prompt }) };
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
// v0.4: deterministic synthetic usage, so a zero-model-call stub run still
|
|
51
|
+
// exercises the whole economics path (usage → pricing snapshot → derived cost/
|
|
52
|
+
// latency fields → the value report's three axes). The numbers are SYNTHETIC and
|
|
53
|
+
// deliberately shaped like the real surfaces: a large fixed harness preamble that
|
|
54
|
+
// is identical in both arms, plus the with-skill arm's extra SKILL.md input and
|
|
55
|
+
// its slightly longer, slower output. No randomness — same input, same usage.
|
|
56
|
+
const STUB_HARNESS_PREAMBLE_TOKENS = 20000; // the fixed, arm-identical overhead
|
|
57
|
+
function stubUsage({ kind, system, prompt }) {
|
|
58
|
+
const promptTokens = Math.ceil(String(prompt || '').length / 4);
|
|
59
|
+
if (kind === 'judge') {
|
|
60
|
+
return { input_tokens: 1200 + promptTokens, output_tokens: 60, cached_tokens: 800, wall_ms: 900 };
|
|
61
|
+
}
|
|
62
|
+
const skillTokens = system ? Math.ceil(String(system).length / 4) : 0;
|
|
63
|
+
return {
|
|
64
|
+
input_tokens: STUB_HARNESS_PREAMBLE_TOKENS + promptTokens + skillTokens,
|
|
65
|
+
output_tokens: system ? 420 : 350,
|
|
66
|
+
cached_tokens: STUB_HARNESS_PREAMBLE_TOKENS,
|
|
67
|
+
wall_ms: system ? 4200 : 3600,
|
|
68
|
+
};
|
|
48
69
|
}
|
|
49
70
|
|
|
50
71
|
function stubEnabled() {
|
|
51
72
|
return process.env.DRIFTPROOF_STUB === '1';
|
|
52
73
|
}
|
|
53
74
|
|
|
54
|
-
module.exports = { stubComplete, stubEnabled, MARK_SKILL, MARK_BASE };
|
|
75
|
+
module.exports = { stubComplete, stubEnabled, stubUsage, MARK_SKILL, MARK_BASE, STUB_HARNESS_PREAMBLE_TOKENS };
|
package/lib/usage.js
ADDED
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// Usage capture — normalizing what each surface reports about a single call.
|
|
5
|
+
//
|
|
6
|
+
// Every lane reports token usage in its own shape, and two of them (the CLI
|
|
7
|
+
// lanes) were previously discarding it entirely: the `claude -p` default text
|
|
8
|
+
// output carries no usage block at all, and `codex exec --json` streams usage on
|
|
9
|
+
// stdout that we were routing to /dev/null. Spec v0.4 captures it.
|
|
10
|
+
//
|
|
11
|
+
// ONE normalized shape, so three substrates are comparable:
|
|
12
|
+
//
|
|
13
|
+
// { input_tokens, output_tokens, cached_tokens, wall_ms }
|
|
14
|
+
//
|
|
15
|
+
// input_tokens TOTAL input presented to the model for this call, INCLUDING
|
|
16
|
+
// any portion served from cache. This is the field that most
|
|
17
|
+
// needs normalizing: the surfaces disagree about it (below).
|
|
18
|
+
// cached_tokens the portion of input_tokens served from cache (null when the
|
|
19
|
+
// surface does not surface it — never guessed, never 0-filled).
|
|
20
|
+
// output_tokens tokens generated, INCLUDING reasoning/thinking tokens where
|
|
21
|
+
// the surface bundles them (both CLI lanes do; the split is
|
|
22
|
+
// reported by the surface but deliberately not modelled here —
|
|
23
|
+
// see the disclosure in REPORT-STYLE.md).
|
|
24
|
+
// wall_ms measured BY US around the call (see lib/provider), not taken
|
|
25
|
+
// from the surface. It is the only field whose definition is
|
|
26
|
+
// identical across lanes, which is exactly why latency is
|
|
27
|
+
// reported as an observed, surface-disclosed number.
|
|
28
|
+
//
|
|
29
|
+
// THE INPUT-TOKEN DISAGREEMENT (verified against real output, 2026-08-18):
|
|
30
|
+
// - `claude -p --output-format json` reports input_tokens EXCLUDING cache:
|
|
31
|
+
// total input = input_tokens + cache_creation_input_tokens + cache_read_input_tokens.
|
|
32
|
+
// (Observed: 10 + 6,974 + 18,134 = 25,118 for a two-word prompt.)
|
|
33
|
+
// - `codex exec --json` reports input_tokens INCLUDING cache, with
|
|
34
|
+
// cached_input_tokens as a SUBSET of it. (Observed: 10,807 of which 4,480 cached.)
|
|
35
|
+
// Normalizing to "total including cache" makes the two comparable; the raw
|
|
36
|
+
// fixtures both parsers are tested against are checked in under tests/fixtures/.
|
|
37
|
+
//
|
|
38
|
+
// Both CLI surfaces prepend a large fixed harness preamble that we do not
|
|
39
|
+
// control (~25k tokens on claude-cli, ~11k on codex). It is IDENTICAL in the
|
|
40
|
+
// with-skill and baseline arms, so it cancels in the incremental (Δ) figures —
|
|
41
|
+
// which is why Δcost is the honest axis and absolute per-call cost is not.
|
|
42
|
+
|
|
43
|
+
function n(x) { return Number.isFinite(Number(x)) ? Number(x) : 0; }
|
|
44
|
+
|
|
45
|
+
// The empty/unknown usage record. Deliberately null (not zeros) so "the surface
|
|
46
|
+
// did not tell us" never reads as "the call cost nothing".
|
|
47
|
+
function emptyUsage() {
|
|
48
|
+
return { input_tokens: null, output_tokens: null, cached_tokens: null, wall_ms: null };
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
function hasUsage(u) {
|
|
52
|
+
return !!(u && (u.input_tokens != null || u.output_tokens != null));
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
// ── anthropic/cli — `claude -p --output-format json` ──────────────────────────
|
|
56
|
+
// The whole call is one JSON object; `result` carries the final text and `usage`
|
|
57
|
+
// the token counts. input_tokens EXCLUDES cache, so both cache fields are added
|
|
58
|
+
// back to reach the total actually presented to the model.
|
|
59
|
+
function parseClaudeCliJson(stdout) {
|
|
60
|
+
let j;
|
|
61
|
+
try { j = JSON.parse(String(stdout || '')); } catch (_e) { return null; }
|
|
62
|
+
if (!j || typeof j !== 'object') return null;
|
|
63
|
+
const u = j.usage || {};
|
|
64
|
+
const cacheRead = n(u.cache_read_input_tokens);
|
|
65
|
+
const cacheCreate = n(u.cache_creation_input_tokens);
|
|
66
|
+
const usage = {
|
|
67
|
+
input_tokens: n(u.input_tokens) + cacheRead + cacheCreate,
|
|
68
|
+
output_tokens: n(u.output_tokens),
|
|
69
|
+
// Cache READ is the portion genuinely served from cache. Cache CREATION was
|
|
70
|
+
// processed fresh this call (and billed at a premium), so it is not cached.
|
|
71
|
+
cached_tokens: cacheRead,
|
|
72
|
+
wall_ms: null,
|
|
73
|
+
};
|
|
74
|
+
return {
|
|
75
|
+
text: typeof j.result === 'string' ? j.result : '',
|
|
76
|
+
usage,
|
|
77
|
+
isError: j.is_error === true,
|
|
78
|
+
// The CLI reports its own dollar figure. Recorded here for completeness but
|
|
79
|
+
// NOT used: costs are computed uniformly from the frozen pricing snapshot so
|
|
80
|
+
// three substrates are on one basis (see lib/value.js).
|
|
81
|
+
surfaceReportedCostUsd: Number.isFinite(Number(j.total_cost_usd)) ? Number(j.total_cost_usd) : null,
|
|
82
|
+
};
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
// ── openai/cli — `codex exec --json` JSONL event stream ───────────────────────
|
|
86
|
+
// One JSON object per line. Usage rides the terminal `turn.completed` event;
|
|
87
|
+
// input_tokens already INCLUDES cached_input_tokens. Unparseable lines are
|
|
88
|
+
// skipped (the stream also carries progress events we do not model).
|
|
89
|
+
function parseCodexJsonl(stdout) {
|
|
90
|
+
const lines = String(stdout || '').split('\n').map((l) => l.trim()).filter(Boolean);
|
|
91
|
+
let usage = null;
|
|
92
|
+
let text = '';
|
|
93
|
+
for (const line of lines) {
|
|
94
|
+
let ev;
|
|
95
|
+
try { ev = JSON.parse(line); } catch (_e) { continue; }
|
|
96
|
+
if (!ev || typeof ev !== 'object') continue;
|
|
97
|
+
if (ev.type === 'turn.completed' && ev.usage) {
|
|
98
|
+
const u = ev.usage;
|
|
99
|
+
usage = {
|
|
100
|
+
input_tokens: n(u.input_tokens), // already total (cache included)
|
|
101
|
+
output_tokens: n(u.output_tokens), // includes reasoning_output_tokens
|
|
102
|
+
cached_tokens: u.cached_input_tokens == null ? null : n(u.cached_input_tokens),
|
|
103
|
+
wall_ms: null,
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
// The final message is normally read from the -o file; this is a fallback.
|
|
107
|
+
if (ev.type === 'item.completed' && ev.item && ev.item.type === 'agent_message' && typeof ev.item.text === 'string') {
|
|
108
|
+
text = ev.item.text;
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
return { usage, text };
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
// ── anthropic/api ─────────────────────────────────────────────────────────────
|
|
115
|
+
// The Messages API reports input_tokens EXCLUDING cache, same as the CLI.
|
|
116
|
+
function parseAnthropicApiUsage(u) {
|
|
117
|
+
if (!u) return emptyUsage();
|
|
118
|
+
const cacheRead = n(u.cache_read_input_tokens);
|
|
119
|
+
const cacheCreate = n(u.cache_creation_input_tokens);
|
|
120
|
+
return {
|
|
121
|
+
input_tokens: n(u.input_tokens) + cacheRead + cacheCreate,
|
|
122
|
+
output_tokens: n(u.output_tokens),
|
|
123
|
+
cached_tokens: u.cache_read_input_tokens == null ? null : cacheRead,
|
|
124
|
+
wall_ms: null,
|
|
125
|
+
};
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
// ── openai/api (Chat Completions-compatible) ──────────────────────────────────
|
|
129
|
+
// prompt_tokens is the total; the cached portion, when present, is nested under
|
|
130
|
+
// prompt_tokens_details.cached_tokens.
|
|
131
|
+
function parseOpenaiApiUsage(u) {
|
|
132
|
+
if (!u) return emptyUsage();
|
|
133
|
+
const details = u.prompt_tokens_details || {};
|
|
134
|
+
return {
|
|
135
|
+
input_tokens: n(u.prompt_tokens),
|
|
136
|
+
output_tokens: n(u.completion_tokens),
|
|
137
|
+
cached_tokens: details.cached_tokens == null ? null : n(details.cached_tokens),
|
|
138
|
+
wall_ms: null,
|
|
139
|
+
};
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
// Sum a list of usage records into one (used for the N judge samples of a case).
|
|
143
|
+
// Null-safe: unknown fields stay null unless at least one record reported them.
|
|
144
|
+
function sumUsage(list) {
|
|
145
|
+
const records = (list || []).filter(Boolean);
|
|
146
|
+
if (!records.length) return emptyUsage();
|
|
147
|
+
const out = { input_tokens: null, output_tokens: null, cached_tokens: null, wall_ms: null };
|
|
148
|
+
for (const key of ['input_tokens', 'output_tokens', 'cached_tokens', 'wall_ms']) {
|
|
149
|
+
const present = records.filter((r) => r[key] != null);
|
|
150
|
+
if (present.length) out[key] = present.reduce((a, r) => a + n(r[key]), 0);
|
|
151
|
+
}
|
|
152
|
+
return out;
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
// Round a usage record's fields to integers (tokens and ms are whole units).
|
|
156
|
+
function normalizeUsage(u) {
|
|
157
|
+
if (!u) return emptyUsage();
|
|
158
|
+
const out = emptyUsage();
|
|
159
|
+
for (const key of ['input_tokens', 'output_tokens', 'cached_tokens', 'wall_ms']) {
|
|
160
|
+
if (u[key] != null && Number.isFinite(Number(u[key]))) out[key] = Math.round(Number(u[key]));
|
|
161
|
+
}
|
|
162
|
+
return out;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
module.exports = {
|
|
166
|
+
emptyUsage, hasUsage, sumUsage, normalizeUsage,
|
|
167
|
+
parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage,
|
|
168
|
+
};
|