driftproof 0.4.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +72 -17
- package/bin/driftproof +28 -5
- package/config.js +18 -2
- package/lib/diff.js +58 -7
- package/lib/hygiene.js +113 -0
- package/lib/judge.js +11 -3
- package/lib/provider.js +55 -25
- package/lib/receipt.js +8 -2
- package/lib/revision.js +162 -0
- package/lib/run.js +32 -5
- package/lib/stub.js +24 -3
- package/lib/usage.js +168 -0
- package/lib/value.js +502 -0
- package/package.json +2 -2
- package/spec/RECEIPT.md +52 -9
- package/spec/receipt.schema.json +703 -76
- package/spec/receipt.v0.3.1.schema.json +642 -0
package/lib/provider.js
CHANGED
|
@@ -7,6 +7,9 @@ const path = require('path');
|
|
|
7
7
|
const { spawn } = require('child_process');
|
|
8
8
|
const { withRetry, withTimeout } = require('./json');
|
|
9
9
|
const { stubComplete, stubEnabled } = require('./stub');
|
|
10
|
+
const {
|
|
11
|
+
parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage, normalizeUsage,
|
|
12
|
+
} = require('./usage');
|
|
10
13
|
|
|
11
14
|
// Provider abstraction: one `complete()` call over a TWO-AXIS lane model —
|
|
12
15
|
// provider (anthropic | openai) × surface (api | cli). Four concrete lanes:
|
|
@@ -109,9 +112,18 @@ const CODEX_OVERHEAD_NOTE =
|
|
|
109
112
|
+ 'control; the model id is set by us via -m (it is not echoed in the JSONL stream). '
|
|
110
113
|
+ 'Approval prompting is off by default on `codex exec` (no -a flag is passed).';
|
|
111
114
|
|
|
112
|
-
// Send a single-turn prompt and return { text, usage, surface, provider }.
|
|
113
|
-
//
|
|
114
|
-
//
|
|
115
|
+
// Send a single-turn prompt and return { text, usage, wall_ms, surface, provider }.
|
|
116
|
+
//
|
|
117
|
+
// usage is the normalized v0.4 record { input_tokens, output_tokens, cached_tokens,
|
|
118
|
+
// wall_ms } on EVERY lane — including the two CLI lanes, which report it in their
|
|
119
|
+
// structured output (`claude -p --output-format json`; the `codex exec --json`
|
|
120
|
+
// JSONL stream). See lib/usage.js for the per-surface shapes and the input-token
|
|
121
|
+
// normalization. Fields the surface does not report stay null, never 0.
|
|
122
|
+
//
|
|
123
|
+
// wall_ms is measured HERE, around the successful attempt, so it means the same
|
|
124
|
+
// thing on all four lanes (retries are excluded — a retried call's latency would
|
|
125
|
+
// describe our backoff, not the model).
|
|
126
|
+
//
|
|
115
127
|
// `temperature` is honoured ONLY on api surfaces; cli surfaces control sampling
|
|
116
128
|
// themselves, so temperature is ignored there and the receipt records that fact.
|
|
117
129
|
async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = null, temperature = undefined }) {
|
|
@@ -120,7 +132,10 @@ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = n
|
|
|
120
132
|
// Offline stub surface: canned completion, zero model calls. The receipt still
|
|
121
133
|
// records the real surface/provider so a stub run is not mistaken for a genuine
|
|
122
134
|
// one at read time — only the generation/judge TEXT is canned.
|
|
123
|
-
if (stubEnabled())
|
|
135
|
+
if (stubEnabled()) {
|
|
136
|
+
const s = stubComplete({ system, prompt });
|
|
137
|
+
return { ...s, usage: normalizeUsage(s.usage), wall_ms: (s.usage && s.usage.wall_ms) || 0, surface, provider, attempts: 1 };
|
|
138
|
+
}
|
|
124
139
|
|
|
125
140
|
// Per-surface tolerance (see retryPolicyForSurface): a cli/subscription surface
|
|
126
141
|
// spawns a first-party CLI subprocess whose cold-start is slow and occasionally
|
|
@@ -144,10 +159,20 @@ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = n
|
|
|
144
159
|
// throttling/cold-starts so one blip doesn't abort a multi-hour grind. `attempts`
|
|
145
160
|
// counts every try (retries included) so the budget can charge for them.
|
|
146
161
|
let attempts = 0;
|
|
147
|
-
|
|
162
|
+
// Wall-clock of the attempt that SUCCEEDED (each attempt overwrites, so a
|
|
163
|
+
// retried call reports the latency of the call that actually produced the text,
|
|
164
|
+
// not the accumulated backoff).
|
|
165
|
+
let wallMs = null;
|
|
166
|
+
const runner = () => {
|
|
167
|
+
attempts += 1;
|
|
168
|
+
const t0 = Date.now();
|
|
169
|
+
return withTimeout(laneRunner, effTimeout, `provider(${surface})`)
|
|
170
|
+
.then((r) => { wallMs = Date.now() - t0; return r; });
|
|
171
|
+
};
|
|
148
172
|
try {
|
|
149
173
|
const out = await withRetry(runner, { tries: policy.tries, baseDelayMs });
|
|
150
|
-
|
|
174
|
+
const usage = normalizeUsage({ ...(out.usage || {}), wall_ms: wallMs });
|
|
175
|
+
return { ...out, usage, wall_ms: wallMs, surface, provider, attempts };
|
|
151
176
|
} catch (e) {
|
|
152
177
|
// Surface the attempt count so a persistently-failing call can be charged for
|
|
153
178
|
// (and, for a timeout, marked failed_timeout by the runner instead of fatal).
|
|
@@ -172,13 +197,7 @@ async function completeAnthropicApi({ system, prompt, model, maxTokens, temperat
|
|
|
172
197
|
if (temperature !== undefined) params.temperature = temperature;
|
|
173
198
|
const resp = await client.messages.create(params);
|
|
174
199
|
const text = (resp.content || []).filter((b) => b.type === 'text').map((b) => b.text).join('');
|
|
175
|
-
return {
|
|
176
|
-
text,
|
|
177
|
-
usage: {
|
|
178
|
-
input_tokens: (resp.usage && resp.usage.input_tokens) || 0,
|
|
179
|
-
output_tokens: (resp.usage && resp.usage.output_tokens) || 0,
|
|
180
|
-
},
|
|
181
|
-
};
|
|
200
|
+
return { text, usage: parseAnthropicApiUsage(resp.usage) };
|
|
182
201
|
}
|
|
183
202
|
|
|
184
203
|
// ── anthropic/cli (claude -p) ──────────────────────────────────────────────────
|
|
@@ -189,7 +208,13 @@ function completeClaudeCli({ system, prompt, model, timeoutMs }) {
|
|
|
189
208
|
const env = { ...process.env };
|
|
190
209
|
delete env.ANTHROPIC_API_KEY;
|
|
191
210
|
|
|
192
|
-
|
|
211
|
+
// v0.4: `--output-format json` returns ONE JSON object carrying both the
|
|
212
|
+
// final text (`result`) and the token usage (`usage`) — the default text
|
|
213
|
+
// output carries no usage at all, which is why usage was previously null on
|
|
214
|
+
// this lane. The text is read from the parsed object; if the CLI ever emits
|
|
215
|
+
// something unparseable we fall back to the raw stdout so a run degrades to
|
|
216
|
+
// the old behaviour (text, no usage) rather than failing.
|
|
217
|
+
const args = ['-p', '--output-format', 'json', '--model', resolveModel(model)];
|
|
193
218
|
if (system) args.push('--append-system-prompt', system);
|
|
194
219
|
|
|
195
220
|
const child = spawn('claude', args, { env, stdio: ['pipe', 'pipe', 'pipe'] });
|
|
@@ -206,7 +231,10 @@ function completeClaudeCli({ system, prompt, model, timeoutMs }) {
|
|
|
206
231
|
child.on('close', (code) => {
|
|
207
232
|
clearTimeout(killer);
|
|
208
233
|
if (code !== 0) return reject(new Error(`claude CLI exited ${code}: ${err.slice(0, 400)}`));
|
|
209
|
-
|
|
234
|
+
const parsed = parseClaudeCliJson(out);
|
|
235
|
+
if (!parsed) return resolve({ text: out.trim(), usage: null });
|
|
236
|
+
if (parsed.isError) return reject(new Error(`claude CLI reported is_error: ${String(parsed.text || '').slice(0, 300)}`));
|
|
237
|
+
resolve({ text: String(parsed.text || '').trim(), usage: parsed.usage });
|
|
210
238
|
});
|
|
211
239
|
child.stdin.write(prompt);
|
|
212
240
|
child.stdin.end();
|
|
@@ -250,14 +278,7 @@ async function completeOpenaiApi({ system, prompt, model, maxTokens, temperature
|
|
|
250
278
|
let parsed;
|
|
251
279
|
try { parsed = JSON.parse(raw); } catch (_e) { throw new Error(`openai/api: unparseable response: ${raw.slice(0, 200)}`); }
|
|
252
280
|
const text = ((parsed.choices && parsed.choices[0] && parsed.choices[0].message && parsed.choices[0].message.content) || '');
|
|
253
|
-
|
|
254
|
-
return {
|
|
255
|
-
text: String(text).trim(),
|
|
256
|
-
usage: {
|
|
257
|
-
input_tokens: usage.prompt_tokens || 0,
|
|
258
|
-
output_tokens: usage.completion_tokens || 0,
|
|
259
|
-
},
|
|
260
|
-
};
|
|
281
|
+
return { text: String(text).trim(), usage: parseOpenaiApiUsage(parsed.usage) };
|
|
261
282
|
}
|
|
262
283
|
|
|
263
284
|
// Read the OpenAI provider config (base_url + api-key env) from the registry,
|
|
@@ -325,9 +346,15 @@ function completeCodexCli({ system, prompt, model, timeoutMs }) {
|
|
|
325
346
|
// stdin (per its --help), which is content-agnostic.
|
|
326
347
|
const args = codexFinalArgs({ model, outFile });
|
|
327
348
|
|
|
328
|
-
|
|
349
|
+
// v0.4: stdout is CAPTURED (it was 'ignore'). `--json` streams JSONL events
|
|
350
|
+
// there, and the terminal `turn.completed` event carries this call's token
|
|
351
|
+
// usage — the only place codex reports it. The final message still comes from
|
|
352
|
+
// the -o file (cleaner than scraping the stream); the JSONL is read for usage.
|
|
353
|
+
const child = spawn('codex', args, { env: { ...process.env }, stdio: ['pipe', 'pipe', 'pipe'] });
|
|
329
354
|
let err = '';
|
|
355
|
+
let jsonl = '';
|
|
330
356
|
const killer = setTimeout(() => { child.kill('SIGKILL'); }, timeoutMs + 5000);
|
|
357
|
+
child.stdout.on('data', (d) => { jsonl += d; });
|
|
331
358
|
child.stderr.on('data', (d) => { err += d; });
|
|
332
359
|
child.on('error', (e) => {
|
|
333
360
|
clearTimeout(killer);
|
|
@@ -340,7 +367,10 @@ function completeCodexCli({ system, prompt, model, timeoutMs }) {
|
|
|
340
367
|
try { text = fs.readFileSync(outFile, 'utf8'); } catch (_e) { /* fall through */ }
|
|
341
368
|
try { fs.unlinkSync(outFile); } catch (_e) { /* best effort */ }
|
|
342
369
|
if (code !== 0) return reject(new Error(`codex exec exited ${code}: ${String(err).slice(0, 400)}`));
|
|
343
|
-
|
|
370
|
+
const ev = parseCodexJsonl(jsonl);
|
|
371
|
+
// Prefer the -o file; fall back to the stream's agent_message if it is empty.
|
|
372
|
+
const finalText = String(text || ev.text || '').trim();
|
|
373
|
+
resolve({ text: finalText, usage: ev.usage });
|
|
344
374
|
});
|
|
345
375
|
child.stdin.on('error', () => { /* ignore EPIPE if codex exits before reading all input */ });
|
|
346
376
|
child.stdin.write(fullPrompt);
|
package/lib/receipt.js
CHANGED
|
@@ -14,7 +14,8 @@ const SCHEMA_FILES = {
|
|
|
14
14
|
'0.1': 'receipt.v0.1.schema.json',
|
|
15
15
|
'0.2': 'receipt.v0.2.schema.json',
|
|
16
16
|
'0.3': 'receipt.v0.3.schema.json',
|
|
17
|
-
'0.3.1': 'receipt.schema.json',
|
|
17
|
+
'0.3.1': 'receipt.v0.3.1.schema.json',
|
|
18
|
+
'0.4': 'receipt.schema.json',
|
|
18
19
|
};
|
|
19
20
|
|
|
20
21
|
const _validators = {};
|
|
@@ -80,7 +81,7 @@ function aggregate(caseResults) {
|
|
|
80
81
|
// run: { model_id, model_release_date, surface, runner_version, date_utc, judge, registry, transcripts }
|
|
81
82
|
// cases: [ { id, mode, outcome, score, mean, stddev, samples, generation_hash, judge_sample_hashes, threshold, reason, judge } ]
|
|
82
83
|
// editorialReviews: optional [ { url, source, date } ]
|
|
83
|
-
function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED', editorialReviews = null }) {
|
|
84
|
+
function buildReceipt({ skill, suite, run, cases, economics = null, verificationLevel = 'TESTED', editorialReviews = null }) {
|
|
84
85
|
// v0.3.1: cases marked failed_timeout are recorded in results.cases but EXCLUDED
|
|
85
86
|
// from aggregates — a band is never fabricated from a case that did not complete.
|
|
86
87
|
const okCases = cases.filter((c) => c.case_status !== 'failed_timeout');
|
|
@@ -138,6 +139,11 @@ function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED',
|
|
|
138
139
|
// skill.tokens — estimated SKILL.md token size (value-per-token axis).
|
|
139
140
|
if (run.surface_overhead_note) receipt.run.surface_overhead_note = run.surface_overhead_note;
|
|
140
141
|
if (skill.tokens != null) receipt.skill.tokens = skill.tokens;
|
|
142
|
+
// v0.4 economics (additive-optional): the frozen prices this receipt's derived
|
|
143
|
+
// dollar figures were computed from, and the derived block itself. A receipt
|
|
144
|
+
// from a surface that reports no usage simply omits both.
|
|
145
|
+
if (run.pricing_snapshot) receipt.run.pricing_snapshot = run.pricing_snapshot;
|
|
146
|
+
if (economics) receipt.economics = economics;
|
|
141
147
|
// v0.3.1: mark the receipt incomplete when any case failed (excluded above).
|
|
142
148
|
if (failedCount > 0) {
|
|
143
149
|
receipt.run.status = 'incomplete';
|
package/lib/revision.js
ADDED
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
const { bandVerdict, round } = require('./stats');
|
|
5
|
+
const { EFFECT_FLOOR } = require('../config');
|
|
6
|
+
|
|
7
|
+
// Revision drift — the fifth report type (spec 009, Report #006).
|
|
8
|
+
//
|
|
9
|
+
// Every other report type holds the skill text fixed and moves something
|
|
10
|
+
// underneath it: the model release (#001, #003), the vendor surface (#002), the
|
|
11
|
+
// capability tier (#004), the axes and the price (#005). This one inverts the
|
|
12
|
+
// design. The substrate is held still — same model, same provider, same surface,
|
|
13
|
+
// same suite, same fixed judge, same sampling — and the SKILL'S OWN TEXT moves,
|
|
14
|
+
// from the revision a report pinned to the revision upstream ships today.
|
|
15
|
+
//
|
|
16
|
+
// This module holds the report type's LANGUAGE as code rather than as
|
|
17
|
+
// hand-written page copy. That is deliberate: the fairness rule below is the one
|
|
18
|
+
// a measurement project is most tempted to apply in one direction only, and a
|
|
19
|
+
// rule that lives in prose cannot be gated before the run that would tempt it.
|
|
20
|
+
|
|
21
|
+
// ── the cell headline ────────────────────────────────────────────────────────
|
|
22
|
+
// A summary of the per-case band-overlap verdicts, worded about the REVISION.
|
|
23
|
+
// The release-drift headline says "the skill is measurably weaker", which is a
|
|
24
|
+
// sentence about a skill under a moving model. Here the model is the control.
|
|
25
|
+
function revisionHeadline(perCase) {
|
|
26
|
+
const reg = perCase.filter((r) => r.verdict === 'regression').length;
|
|
27
|
+
const imp = perCase.filter((r) => r.verdict === 'improvement').length;
|
|
28
|
+
const s = (n) => (n === 1 ? '' : 's');
|
|
29
|
+
if (reg && imp) {
|
|
30
|
+
return `MIXED — the revision improved ${imp} case${s(imp)} and regressed ${reg} on non-overlapping bands.`;
|
|
31
|
+
}
|
|
32
|
+
if (reg) {
|
|
33
|
+
return `REVISION REGRESSED — ${reg} case${s(reg)} scored lower under the current upstream text (bands do not overlap).`;
|
|
34
|
+
}
|
|
35
|
+
if (imp) {
|
|
36
|
+
return `REVISION IMPROVED — ${imp} case${s(imp)} scored higher under the current upstream text (bands do not overlap); none regressed.`;
|
|
37
|
+
}
|
|
38
|
+
return 'WITHIN NOISE — the revision moved no case beyond its confidence band; the pinned text and the current text measure the same.';
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
// Classification word for a cell, from its headline. Kept separate so a caller
|
|
42
|
+
// can branch on the class without parsing prose.
|
|
43
|
+
function revisionClass(perCase) {
|
|
44
|
+
const h = revisionHeadline(perCase);
|
|
45
|
+
return h.split(' —')[0];
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
// ── the fairness sentence ────────────────────────────────────────────────────
|
|
49
|
+
// Spec 009 § Fairness, clauses 1 and 4. Where a revision IMPROVED a skill, the
|
|
50
|
+
// published #005 figure understates the pack a reader can install today; where it
|
|
51
|
+
// REGRESSED one, #005 overstates it. Both sentences are generated by the same
|
|
52
|
+
// function, from the same template, so the disclosure cannot quietly become a
|
|
53
|
+
// one-directional courtesy — the symmetry is a property of the code, and the
|
|
54
|
+
// gate asserts it.
|
|
55
|
+
//
|
|
56
|
+
// A cell within noise gets NO sentence. #005's figure stands unamended, because
|
|
57
|
+
// nothing was measured that would amend it, and manufacturing a hedge for a null
|
|
58
|
+
// result is how a report launders noise into a finding.
|
|
59
|
+
function fairnessSentence({ slug, classification, report005Delta, measuredDelta }) {
|
|
60
|
+
const cls = String(classification || '');
|
|
61
|
+
if (cls !== 'REVISION IMPROVED' && cls !== 'REVISION REGRESSED') return null;
|
|
62
|
+
const improved = cls === 'REVISION IMPROVED';
|
|
63
|
+
const direction = improved ? 'understates' : 'overstates';
|
|
64
|
+
const d = (n) => (n == null ? 'n/a' : (n >= 0 ? '+' : '') + Number(n).toFixed(3));
|
|
65
|
+
return `Report #005 measured ${slug} at ${d(report005Delta)} on the text it had pinned. `
|
|
66
|
+
+ `Report #006 measures the current upstream revision at ${d(measuredDelta)} on the same substrate and the same suite. `
|
|
67
|
+
+ `#005's published figure therefore ${direction} the pack upstream ships today for this skill, `
|
|
68
|
+
+ `and is amended by this report rather than corrected in place.`;
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
// ── per-cell scoping disclosures ─────────────────────────────────────────────
|
|
72
|
+
// Spec 009 AC-10. One cell in Report #006 measures something other than what its
|
|
73
|
+
// upstream author changed it to do, and the reader looking at that row is the
|
|
74
|
+
// reader who needs to be told.
|
|
75
|
+
const SCOPING_NOTES = {
|
|
76
|
+
'git-workflow-and-versioning':
|
|
77
|
+
'This revision changes the frontmatter `description:` line. In a skill runtime a description is a '
|
|
78
|
+
+ 'routing trigger: it decides whether the skill loads, and never reaches the model as guidance. '
|
|
79
|
+
+ 'Driftproof makes no routing decision — it always injects the skill, and passes the whole file, '
|
|
80
|
+
+ 'frontmatter included, as the system prompt. This cell therefore measures the revision as added '
|
|
81
|
+
+ 'context and cannot measure it as a trigger.',
|
|
82
|
+
};
|
|
83
|
+
function scopingNote(slug) {
|
|
84
|
+
return SCOPING_NOTES[slug] || null;
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
// ── the baseline-reproduction control ────────────────────────────────────────
|
|
88
|
+
// Spec 009 AC-6, and the thing that makes the free pinned arm honest.
|
|
89
|
+
//
|
|
90
|
+
// Report #006 reuses #005's receipts as the pinned-text arm. That is valid only
|
|
91
|
+
// if the substrate has not moved, and `run.model_release_date` is null on every
|
|
92
|
+
// #005 receipt, so id equality is the only version evidence a receipt carries. A
|
|
93
|
+
// provider that re-points a concrete id at a new snapshot is invisible to it.
|
|
94
|
+
//
|
|
95
|
+
// It does not have to be. Every fresh run emits a BASELINE arm: the same cases,
|
|
96
|
+
// the same substrate, and no skill text at all. The revision cannot touch it by
|
|
97
|
+
// construction, so comparing the fresh baseline against the reused receipt's
|
|
98
|
+
// baseline re-measures exactly the thing id equality could not prove — at no
|
|
99
|
+
// extra cost, because that arm is already paid for.
|
|
100
|
+
//
|
|
101
|
+
// A cell whose baselines do not reproduce is NOT MEASURED. The reuse is a tested
|
|
102
|
+
// prediction, not an assumption the report asks the reader to grant.
|
|
103
|
+
function baselineBands(receipt) {
|
|
104
|
+
const out = {};
|
|
105
|
+
for (const c of receipt.results.cases) {
|
|
106
|
+
if (c.mode !== 'baseline') continue;
|
|
107
|
+
out[c.id] = { mean: c.mean != null ? c.mean : c.score, stddev: c.stddev || 0 };
|
|
108
|
+
}
|
|
109
|
+
return out;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
function baselineControl(reused, fresh) {
|
|
113
|
+
const A = baselineBands(reused);
|
|
114
|
+
const B = baselineBands(fresh);
|
|
115
|
+
const ids = [...new Set([...Object.keys(A), ...Object.keys(B)])];
|
|
116
|
+
|
|
117
|
+
const perCase = ids.map((id) => {
|
|
118
|
+
const before = A[id] || null;
|
|
119
|
+
const after = B[id] || null;
|
|
120
|
+
if (!before || !after) return { id, before, after, delta: null, moved: false, missing: true };
|
|
121
|
+
const delta = round(after.mean - before.mean);
|
|
122
|
+
// The same rule the study uses everywhere else: band separation AND the
|
|
123
|
+
// effect floor. A baseline that wobbles inside its band has not moved.
|
|
124
|
+
const raw = bandVerdict(before.mean, before.stddev, after.mean, after.stddev);
|
|
125
|
+
const separated = raw === 'regression' || raw === 'improvement';
|
|
126
|
+
return { id, before, after, delta, moved: separated && Math.abs(delta) >= EFFECT_FLOOR, missing: false };
|
|
127
|
+
});
|
|
128
|
+
|
|
129
|
+
const movedCases = perCase.filter((r) => r.moved);
|
|
130
|
+
const missing = perCase.filter((r) => r.missing);
|
|
131
|
+
const reproduced = movedCases.length === 0 && missing.length === 0;
|
|
132
|
+
const aggDelta = round(
|
|
133
|
+
(fresh.comparison && fresh.comparison.baseline_score != null ? fresh.comparison.baseline_score : 0)
|
|
134
|
+
- (reused.comparison && reused.comparison.baseline_score != null ? reused.comparison.baseline_score : 0),
|
|
135
|
+
);
|
|
136
|
+
|
|
137
|
+
return {
|
|
138
|
+
reproduced,
|
|
139
|
+
blocked: !reproduced,
|
|
140
|
+
verdict: reproduced ? 'MEASURED' : 'NOT MEASURED',
|
|
141
|
+
moved_cases: movedCases.map((r) => r.id),
|
|
142
|
+
missing_cases: missing.map((r) => r.id),
|
|
143
|
+
aggregate_baseline_delta: aggDelta,
|
|
144
|
+
floor: EFFECT_FLOOR,
|
|
145
|
+
// THE CONTROL PROVES NON-REPRODUCTION. IT CANNOT SAY WHY. These strings used
|
|
146
|
+
// to read 'the substrate moved' and 'the substrate held still' — a cause,
|
|
147
|
+
// asserted by a comparison that measures two baseline arms and nothing else.
|
|
148
|
+
// A 120-call stability probe then found generation-level sampling noise large
|
|
149
|
+
// enough to account for every gap this control saw, with no substrate movement
|
|
150
|
+
// required, and the report page retracted the claim while three committed
|
|
151
|
+
// control records still carried it (approval finding F-009-N). Reason strings
|
|
152
|
+
// only: no score, sample, hash or verdict changed with this edit.
|
|
153
|
+
reason: reproduced
|
|
154
|
+
? 'the fresh baseline reproduces the reused receipt\'s baseline within the band and the floor, so the pinned-arm reuse stands for this cell'
|
|
155
|
+
: `the fresh baseline does not reproduce the reused receipt's baseline (${movedCases.length} case(s) moved beyond the band and the ${EFFECT_FLOOR} floor${missing.length ? `, ${missing.length} case(s) absent on one side` : ''}) — the reused pinned arm is not comparable to the fresh arm, so revision drift cannot be separated from whatever else changed in this cell; the control establishes non-reproduction and does not identify a cause`,
|
|
156
|
+
perCase,
|
|
157
|
+
};
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
module.exports = {
|
|
161
|
+
revisionHeadline, revisionClass, fairnessSentence, scopingNote, baselineControl,
|
|
162
|
+
};
|
package/lib/run.js
CHANGED
|
@@ -1,14 +1,16 @@
|
|
|
1
1
|
// SPDX-License-Identifier: Apache-2.0
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
-
const { complete, resolveModel, surfaceForModel, CODEX_OVERHEAD_NOTE } = require('./provider');
|
|
4
|
+
const { complete, resolveModel, surfaceForModel, isMeteredSurface, CODEX_OVERHEAD_NOTE } = require('./provider');
|
|
5
5
|
const { gradeSamples, judgeSettings } = require('./judge');
|
|
6
6
|
const { buildReceipt } = require('./receipt');
|
|
7
7
|
const { sha256 } = require('./canonical');
|
|
8
|
-
const { registryStatus, providerForModel } = require('./models');
|
|
8
|
+
const { registryStatus, providerForModel, priceForModel } = require('./models');
|
|
9
9
|
const { perCallCostUSD } = require('./cost');
|
|
10
10
|
const { runChecks } = require('./checks');
|
|
11
11
|
const { estimateTokens } = require('./skillCost');
|
|
12
|
+
const { hasUsage, normalizeUsage } = require('./usage');
|
|
13
|
+
const { buildPricingSnapshot, computeEconomics } = require('./value');
|
|
12
14
|
const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES } = require('../config');
|
|
13
15
|
|
|
14
16
|
// Known model release dates (best-effort; null when unknown). Recorded into the
|
|
@@ -51,8 +53,8 @@ async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs }) {
|
|
|
51
53
|
throw e;
|
|
52
54
|
}
|
|
53
55
|
const system = withSkill ? skillMd : undefined;
|
|
54
|
-
const { text, usage, attempts } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
|
|
55
|
-
return { text, usage, attempts };
|
|
56
|
+
const { text, usage, wall_ms, attempts } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
|
|
57
|
+
return { text, usage, wall_ms, attempts };
|
|
56
58
|
}
|
|
57
59
|
|
|
58
60
|
// Determine a case outcome from its sampled band and threshold.
|
|
@@ -91,6 +93,8 @@ async function judgeCase({ caseObj, response, generationHash, judgeModel, mode,
|
|
|
91
93
|
// v0.3.1 deterministic post-checks (supplementary; NOT folded into `outcome`).
|
|
92
94
|
const checks = runChecks(response, caseObj.checks);
|
|
93
95
|
if (checks.length) caseResult.checks = checks;
|
|
96
|
+
// v0.4: grading overhead for this case row, kept OUT of the skill-value math.
|
|
97
|
+
if (hasUsage(g.usage)) caseResult.judge_usage = g.usage;
|
|
94
98
|
return { caseResult, sampleTexts: g.sample_texts, attempts: g.attempts };
|
|
95
99
|
}
|
|
96
100
|
|
|
@@ -170,6 +174,9 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
170
174
|
const generationHash = sha256(String(gen.text || ''));
|
|
171
175
|
onProgress({ case: c.id, mode, phase: 'judge', samples });
|
|
172
176
|
const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ct, samples });
|
|
177
|
+
// v0.4: the GENERATION call's usage is the skill-value measurement (the
|
|
178
|
+
// judge's own usage rides separately on judge_usage, above).
|
|
179
|
+
if (hasUsage(gen.usage)) jr.caseResult.usage = normalizeUsage({ ...gen.usage, wall_ms: gen.wall_ms });
|
|
173
180
|
calls += samples;
|
|
174
181
|
// Live budget: count all judge calls (retries included) for this (case, mode).
|
|
175
182
|
if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judgeModel, 'judge'));
|
|
@@ -198,6 +205,24 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
198
205
|
const transcripts = keepTranscripts ? pairs.map((p) => p.transcript) : null;
|
|
199
206
|
|
|
200
207
|
const surface = surfaceForModel(modelId);
|
|
208
|
+
const nowIso = opts.nowIso || new Date().toISOString();
|
|
209
|
+
// v0.4 economics. The pricing snapshot is frozen HERE, at run time, from the
|
|
210
|
+
// registry; every derived dollar figure below is computed from the snapshot and
|
|
211
|
+
// never from the live registry, so this receipt keeps its meaning when prices
|
|
212
|
+
// later change.
|
|
213
|
+
const pricingSnapshot = buildPricingSnapshot({
|
|
214
|
+
models: [modelId, judgeModel],
|
|
215
|
+
lookup: priceForModel,
|
|
216
|
+
nowIso,
|
|
217
|
+
});
|
|
218
|
+
const economics = computeEconomics({
|
|
219
|
+
cases: caseResults,
|
|
220
|
+
modelId,
|
|
221
|
+
judgeModelId: judgeModel,
|
|
222
|
+
pricingSnapshot,
|
|
223
|
+
surface,
|
|
224
|
+
meteredSurface: isMeteredSurface(surface),
|
|
225
|
+
});
|
|
201
226
|
const receipt = buildReceipt({
|
|
202
227
|
skill: {
|
|
203
228
|
name: skill.name, version: skill.version, contentHash: skill.contentHash,
|
|
@@ -213,12 +238,14 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
213
238
|
// v0.3.1: on the openai/cli (codex) surface, record the fixed harness preamble.
|
|
214
239
|
surface_overhead_note: surface === 'openai-cli' ? CODEX_OVERHEAD_NOTE : undefined,
|
|
215
240
|
runner_version: RUNNER_VERSION,
|
|
216
|
-
date_utc:
|
|
241
|
+
date_utc: nowIso,
|
|
217
242
|
registry: registryStatus(modelId),
|
|
218
243
|
transcripts: keepTranscripts ? 'retained-local' : 'hashes-only',
|
|
219
244
|
judge: judgeSettings(samples, judgeModel),
|
|
245
|
+
pricing_snapshot: pricingSnapshot,
|
|
220
246
|
},
|
|
221
247
|
cases: caseResults,
|
|
248
|
+
economics,
|
|
222
249
|
verificationLevel: 'TESTED',
|
|
223
250
|
});
|
|
224
251
|
|
package/lib/stub.js
CHANGED
|
@@ -39,16 +39,37 @@ function stubComplete({ system, prompt }) {
|
|
|
39
39
|
pass: score >= 0.7,
|
|
40
40
|
reason: `stub judge (DRIFTPROOF_STUB): ${helped ? 'with-skill marker present' : 'baseline'}`,
|
|
41
41
|
};
|
|
42
|
-
return { text: JSON.stringify(body), usage:
|
|
42
|
+
return { text: JSON.stringify(body), usage: stubUsage({ kind: 'judge', prompt }) };
|
|
43
43
|
}
|
|
44
44
|
// Generation call: canned, marked by mode.
|
|
45
45
|
const mark = system ? MARK_SKILL : MARK_BASE;
|
|
46
46
|
const text = `${mark}\nfeat(stub): canned offline generation for CI (no model was called)`;
|
|
47
|
-
return { text, usage:
|
|
47
|
+
return { text, usage: stubUsage({ kind: 'gen', system, prompt }) };
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
// v0.4: deterministic synthetic usage, so a zero-model-call stub run still
|
|
51
|
+
// exercises the whole economics path (usage → pricing snapshot → derived cost/
|
|
52
|
+
// latency fields → the value report's three axes). The numbers are SYNTHETIC and
|
|
53
|
+
// deliberately shaped like the real surfaces: a large fixed harness preamble that
|
|
54
|
+
// is identical in both arms, plus the with-skill arm's extra SKILL.md input and
|
|
55
|
+
// its slightly longer, slower output. No randomness — same input, same usage.
|
|
56
|
+
const STUB_HARNESS_PREAMBLE_TOKENS = 20000; // the fixed, arm-identical overhead
|
|
57
|
+
function stubUsage({ kind, system, prompt }) {
|
|
58
|
+
const promptTokens = Math.ceil(String(prompt || '').length / 4);
|
|
59
|
+
if (kind === 'judge') {
|
|
60
|
+
return { input_tokens: 1200 + promptTokens, output_tokens: 60, cached_tokens: 800, wall_ms: 900 };
|
|
61
|
+
}
|
|
62
|
+
const skillTokens = system ? Math.ceil(String(system).length / 4) : 0;
|
|
63
|
+
return {
|
|
64
|
+
input_tokens: STUB_HARNESS_PREAMBLE_TOKENS + promptTokens + skillTokens,
|
|
65
|
+
output_tokens: system ? 420 : 350,
|
|
66
|
+
cached_tokens: STUB_HARNESS_PREAMBLE_TOKENS,
|
|
67
|
+
wall_ms: system ? 4200 : 3600,
|
|
68
|
+
};
|
|
48
69
|
}
|
|
49
70
|
|
|
50
71
|
function stubEnabled() {
|
|
51
72
|
return process.env.DRIFTPROOF_STUB === '1';
|
|
52
73
|
}
|
|
53
74
|
|
|
54
|
-
module.exports = { stubComplete, stubEnabled, MARK_SKILL, MARK_BASE };
|
|
75
|
+
module.exports = { stubComplete, stubEnabled, stubUsage, MARK_SKILL, MARK_BASE, STUB_HARNESS_PREAMBLE_TOKENS };
|
package/lib/usage.js
ADDED
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// Usage capture — normalizing what each surface reports about a single call.
|
|
5
|
+
//
|
|
6
|
+
// Every lane reports token usage in its own shape, and two of them (the CLI
|
|
7
|
+
// lanes) were previously discarding it entirely: the `claude -p` default text
|
|
8
|
+
// output carries no usage block at all, and `codex exec --json` streams usage on
|
|
9
|
+
// stdout that we were routing to /dev/null. Spec v0.4 captures it.
|
|
10
|
+
//
|
|
11
|
+
// ONE normalized shape, so three substrates are comparable:
|
|
12
|
+
//
|
|
13
|
+
// { input_tokens, output_tokens, cached_tokens, wall_ms }
|
|
14
|
+
//
|
|
15
|
+
// input_tokens TOTAL input presented to the model for this call, INCLUDING
|
|
16
|
+
// any portion served from cache. This is the field that most
|
|
17
|
+
// needs normalizing: the surfaces disagree about it (below).
|
|
18
|
+
// cached_tokens the portion of input_tokens served from cache (null when the
|
|
19
|
+
// surface does not surface it — never guessed, never 0-filled).
|
|
20
|
+
// output_tokens tokens generated, INCLUDING reasoning/thinking tokens where
|
|
21
|
+
// the surface bundles them (both CLI lanes do; the split is
|
|
22
|
+
// reported by the surface but deliberately not modelled here —
|
|
23
|
+
// see the disclosure in REPORT-STYLE.md).
|
|
24
|
+
// wall_ms measured BY US around the call (see lib/provider), not taken
|
|
25
|
+
// from the surface. It is the only field whose definition is
|
|
26
|
+
// identical across lanes, which is exactly why latency is
|
|
27
|
+
// reported as an observed, surface-disclosed number.
|
|
28
|
+
//
|
|
29
|
+
// THE INPUT-TOKEN DISAGREEMENT (verified against real output, 2026-08-18):
|
|
30
|
+
// - `claude -p --output-format json` reports input_tokens EXCLUDING cache:
|
|
31
|
+
// total input = input_tokens + cache_creation_input_tokens + cache_read_input_tokens.
|
|
32
|
+
// (Observed: 10 + 6,974 + 18,134 = 25,118 for a two-word prompt.)
|
|
33
|
+
// - `codex exec --json` reports input_tokens INCLUDING cache, with
|
|
34
|
+
// cached_input_tokens as a SUBSET of it. (Observed: 10,807 of which 4,480 cached.)
|
|
35
|
+
// Normalizing to "total including cache" makes the two comparable; the raw
|
|
36
|
+
// fixtures both parsers are tested against are checked in under tests/fixtures/.
|
|
37
|
+
//
|
|
38
|
+
// Both CLI surfaces prepend a large fixed harness preamble that we do not
|
|
39
|
+
// control (~25k tokens on claude-cli, ~11k on codex). It is IDENTICAL in the
|
|
40
|
+
// with-skill and baseline arms, so it cancels in the incremental (Δ) figures —
|
|
41
|
+
// which is why Δcost is the honest axis and absolute per-call cost is not.
|
|
42
|
+
|
|
43
|
+
function n(x) { return Number.isFinite(Number(x)) ? Number(x) : 0; }
|
|
44
|
+
|
|
45
|
+
// The empty/unknown usage record. Deliberately null (not zeros) so "the surface
|
|
46
|
+
// did not tell us" never reads as "the call cost nothing".
|
|
47
|
+
function emptyUsage() {
|
|
48
|
+
return { input_tokens: null, output_tokens: null, cached_tokens: null, wall_ms: null };
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
function hasUsage(u) {
|
|
52
|
+
return !!(u && (u.input_tokens != null || u.output_tokens != null));
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
// ── anthropic/cli — `claude -p --output-format json` ──────────────────────────
|
|
56
|
+
// The whole call is one JSON object; `result` carries the final text and `usage`
|
|
57
|
+
// the token counts. input_tokens EXCLUDES cache, so both cache fields are added
|
|
58
|
+
// back to reach the total actually presented to the model.
|
|
59
|
+
function parseClaudeCliJson(stdout) {
|
|
60
|
+
let j;
|
|
61
|
+
try { j = JSON.parse(String(stdout || '')); } catch (_e) { return null; }
|
|
62
|
+
if (!j || typeof j !== 'object') return null;
|
|
63
|
+
const u = j.usage || {};
|
|
64
|
+
const cacheRead = n(u.cache_read_input_tokens);
|
|
65
|
+
const cacheCreate = n(u.cache_creation_input_tokens);
|
|
66
|
+
const usage = {
|
|
67
|
+
input_tokens: n(u.input_tokens) + cacheRead + cacheCreate,
|
|
68
|
+
output_tokens: n(u.output_tokens),
|
|
69
|
+
// Cache READ is the portion genuinely served from cache. Cache CREATION was
|
|
70
|
+
// processed fresh this call (and billed at a premium), so it is not cached.
|
|
71
|
+
cached_tokens: cacheRead,
|
|
72
|
+
wall_ms: null,
|
|
73
|
+
};
|
|
74
|
+
return {
|
|
75
|
+
text: typeof j.result === 'string' ? j.result : '',
|
|
76
|
+
usage,
|
|
77
|
+
isError: j.is_error === true,
|
|
78
|
+
// The CLI reports its own dollar figure. Recorded here for completeness but
|
|
79
|
+
// NOT used: costs are computed uniformly from the frozen pricing snapshot so
|
|
80
|
+
// three substrates are on one basis (see lib/value.js).
|
|
81
|
+
surfaceReportedCostUsd: Number.isFinite(Number(j.total_cost_usd)) ? Number(j.total_cost_usd) : null,
|
|
82
|
+
};
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
// ── openai/cli — `codex exec --json` JSONL event stream ───────────────────────
|
|
86
|
+
// One JSON object per line. Usage rides the terminal `turn.completed` event;
|
|
87
|
+
// input_tokens already INCLUDES cached_input_tokens. Unparseable lines are
|
|
88
|
+
// skipped (the stream also carries progress events we do not model).
|
|
89
|
+
function parseCodexJsonl(stdout) {
|
|
90
|
+
const lines = String(stdout || '').split('\n').map((l) => l.trim()).filter(Boolean);
|
|
91
|
+
let usage = null;
|
|
92
|
+
let text = '';
|
|
93
|
+
for (const line of lines) {
|
|
94
|
+
let ev;
|
|
95
|
+
try { ev = JSON.parse(line); } catch (_e) { continue; }
|
|
96
|
+
if (!ev || typeof ev !== 'object') continue;
|
|
97
|
+
if (ev.type === 'turn.completed' && ev.usage) {
|
|
98
|
+
const u = ev.usage;
|
|
99
|
+
usage = {
|
|
100
|
+
input_tokens: n(u.input_tokens), // already total (cache included)
|
|
101
|
+
output_tokens: n(u.output_tokens), // includes reasoning_output_tokens
|
|
102
|
+
cached_tokens: u.cached_input_tokens == null ? null : n(u.cached_input_tokens),
|
|
103
|
+
wall_ms: null,
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
// The final message is normally read from the -o file; this is a fallback.
|
|
107
|
+
if (ev.type === 'item.completed' && ev.item && ev.item.type === 'agent_message' && typeof ev.item.text === 'string') {
|
|
108
|
+
text = ev.item.text;
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
return { usage, text };
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
// ── anthropic/api ─────────────────────────────────────────────────────────────
|
|
115
|
+
// The Messages API reports input_tokens EXCLUDING cache, same as the CLI.
|
|
116
|
+
function parseAnthropicApiUsage(u) {
|
|
117
|
+
if (!u) return emptyUsage();
|
|
118
|
+
const cacheRead = n(u.cache_read_input_tokens);
|
|
119
|
+
const cacheCreate = n(u.cache_creation_input_tokens);
|
|
120
|
+
return {
|
|
121
|
+
input_tokens: n(u.input_tokens) + cacheRead + cacheCreate,
|
|
122
|
+
output_tokens: n(u.output_tokens),
|
|
123
|
+
cached_tokens: u.cache_read_input_tokens == null ? null : cacheRead,
|
|
124
|
+
wall_ms: null,
|
|
125
|
+
};
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
// ── openai/api (Chat Completions-compatible) ──────────────────────────────────
|
|
129
|
+
// prompt_tokens is the total; the cached portion, when present, is nested under
|
|
130
|
+
// prompt_tokens_details.cached_tokens.
|
|
131
|
+
function parseOpenaiApiUsage(u) {
|
|
132
|
+
if (!u) return emptyUsage();
|
|
133
|
+
const details = u.prompt_tokens_details || {};
|
|
134
|
+
return {
|
|
135
|
+
input_tokens: n(u.prompt_tokens),
|
|
136
|
+
output_tokens: n(u.completion_tokens),
|
|
137
|
+
cached_tokens: details.cached_tokens == null ? null : n(details.cached_tokens),
|
|
138
|
+
wall_ms: null,
|
|
139
|
+
};
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
// Sum a list of usage records into one (used for the N judge samples of a case).
|
|
143
|
+
// Null-safe: unknown fields stay null unless at least one record reported them.
|
|
144
|
+
function sumUsage(list) {
|
|
145
|
+
const records = (list || []).filter(Boolean);
|
|
146
|
+
if (!records.length) return emptyUsage();
|
|
147
|
+
const out = { input_tokens: null, output_tokens: null, cached_tokens: null, wall_ms: null };
|
|
148
|
+
for (const key of ['input_tokens', 'output_tokens', 'cached_tokens', 'wall_ms']) {
|
|
149
|
+
const present = records.filter((r) => r[key] != null);
|
|
150
|
+
if (present.length) out[key] = present.reduce((a, r) => a + n(r[key]), 0);
|
|
151
|
+
}
|
|
152
|
+
return out;
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
// Round a usage record's fields to integers (tokens and ms are whole units).
|
|
156
|
+
function normalizeUsage(u) {
|
|
157
|
+
if (!u) return emptyUsage();
|
|
158
|
+
const out = emptyUsage();
|
|
159
|
+
for (const key of ['input_tokens', 'output_tokens', 'cached_tokens', 'wall_ms']) {
|
|
160
|
+
if (u[key] != null && Number.isFinite(Number(u[key]))) out[key] = Math.round(Number(u[key]));
|
|
161
|
+
}
|
|
162
|
+
return out;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
module.exports = {
|
|
166
|
+
emptyUsage, hasUsage, sumUsage, normalizeUsage,
|
|
167
|
+
parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage,
|
|
168
|
+
};
|