driftproof 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +63 -19
- package/bin/driftproof +66 -4
- package/config/models.json +14 -4
- package/config.js +39 -5
- package/lib/checks.js +50 -0
- package/lib/diff.js +28 -7
- package/lib/export.js +52 -0
- package/lib/importers.js +207 -0
- package/lib/judge.js +27 -13
- package/lib/models.js +58 -8
- package/lib/provider.js +279 -36
- package/lib/receipt.js +22 -3
- package/lib/run.js +80 -26
- package/lib/skill.js +12 -2
- package/lib/skillCost.js +31 -0
- package/lib/verdict.js +10 -1
- package/package.json +1 -1
- package/spec/RECEIPT.md +72 -7
- package/spec/receipt.schema.json +141 -18
- package/spec/receipt.v0.3.schema.json +211 -0
package/lib/provider.js
CHANGED
|
@@ -1,20 +1,33 @@
|
|
|
1
1
|
// SPDX-License-Identifier: Apache-2.0
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
+
const fs = require('fs');
|
|
5
|
+
const os = require('os');
|
|
6
|
+
const path = require('path');
|
|
4
7
|
const { spawn } = require('child_process');
|
|
5
8
|
const { withRetry, withTimeout } = require('./json');
|
|
6
9
|
const { stubComplete, stubEnabled } = require('./stub');
|
|
7
10
|
|
|
8
|
-
// Provider abstraction: one `complete()` call
|
|
11
|
+
// Provider abstraction: one `complete()` call over a TWO-AXIS lane model —
|
|
12
|
+
// provider (anthropic | openai) × surface (api | cli). Four concrete lanes:
|
|
9
13
|
//
|
|
10
|
-
//
|
|
11
|
-
//
|
|
12
|
-
//
|
|
13
|
-
//
|
|
14
|
-
//
|
|
14
|
+
// anthropic/api → Anthropic Messages API (ANTHROPIC_API_KEY) label "api"
|
|
15
|
+
// anthropic/cli → spawn `claude -p` on the subscription label "claude-cli"
|
|
16
|
+
// openai/api → Chat Completions-compatible API (OPENAI_API_KEY, base_url
|
|
17
|
+
// configurable — generic so a future provider is just a
|
|
18
|
+
// registry entry with a base_url) label "openai-api"
|
|
19
|
+
// openai/cli → spawn `codex exec` on the ChatGPT subscription label "openai-cli"
|
|
20
|
+
//
|
|
21
|
+
// The provider of a call is inferred from the MODEL id (so a run can generate on
|
|
22
|
+
// an OpenAI model while the fixed Haiku judge still runs on Anthropic). The
|
|
23
|
+
// surface for each provider is chosen from the environment:
|
|
24
|
+
//
|
|
25
|
+
// anthropic: CLAUDE_PROVIDER = api | cli (default cli — subscription)
|
|
26
|
+
// openai: OPENAI_SURFACE = api | cli (default: api iff OPENAI_API_KEY
|
|
27
|
+
// is present, else cli/codex)
|
|
15
28
|
//
|
|
16
29
|
// The surface actually used is returned so the runner can stamp it into the
|
|
17
|
-
// receipt (run.surface
|
|
30
|
+
// receipt (run.surface), together with run.provider.
|
|
18
31
|
|
|
19
32
|
// Short model aliases → canonical ids. Kept tiny and explicit; unknown values
|
|
20
33
|
// are passed through verbatim so a full model id always works.
|
|
@@ -24,47 +37,131 @@ const MODEL_ALIASES = {
|
|
|
24
37
|
'sonnet-5': 'claude-sonnet-5',
|
|
25
38
|
'sonnet-4-6': 'claude-sonnet-4-6', // previous Sonnet point release
|
|
26
39
|
opus: 'claude-opus-4-8',
|
|
40
|
+
// OpenAI convenience aliases (full ids always work too).
|
|
41
|
+
gpt: 'gpt-5.6-sol',
|
|
42
|
+
'gpt-flagship': 'gpt-5.6-sol',
|
|
27
43
|
};
|
|
28
44
|
|
|
29
45
|
function resolveModel(m) {
|
|
30
46
|
return MODEL_ALIASES[m] || m;
|
|
31
47
|
}
|
|
32
48
|
|
|
49
|
+
// Infer the provider from a (resolved) model id. Pure and self-contained (no
|
|
50
|
+
// registry require) so provider.js has no cycle with lib/models.js; the registry
|
|
51
|
+
// `provider` field is authoritative where a model is registered (see
|
|
52
|
+
// lib/models.providerForModel), and registered OpenAI ids match this prefix too.
|
|
53
|
+
function inferProvider(modelId) {
|
|
54
|
+
const id = String(resolveModel(modelId) || '').toLowerCase();
|
|
55
|
+
if (/^(gpt-|o[1-9]|chatgpt|codex|text-|davinci|omni)/.test(id)) return 'openai';
|
|
56
|
+
return 'anthropic';
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
// ── surface selection per provider ───────────────────────────────────────────
|
|
60
|
+
function anthropicSurface() {
|
|
61
|
+
return (process.env.CLAUDE_PROVIDER || 'cli').toLowerCase() === 'api' ? 'api' : 'claude-cli';
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
function openaiSurface() {
|
|
65
|
+
const pref = (process.env.OPENAI_SURFACE || '').toLowerCase();
|
|
66
|
+
if (pref === 'api') return 'openai-api';
|
|
67
|
+
if (pref === 'cli' || pref === 'codex') return 'openai-cli';
|
|
68
|
+
// Auto: prefer the metered API when a key is present (published-run default),
|
|
69
|
+
// else fall back to the Codex subscription surface.
|
|
70
|
+
return process.env.OPENAI_API_KEY ? 'openai-api' : 'openai-cli';
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
function surfaceForModel(modelId) {
|
|
74
|
+
return inferProvider(modelId) === 'openai' ? openaiSurface() : anthropicSurface();
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
function providerForSurface(surface) {
|
|
78
|
+
return (surface === 'openai-api' || surface === 'openai-cli') ? 'openai' : 'anthropic';
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
// Backward-compatible default surface label (the Anthropic axis). Kept so callers
|
|
82
|
+
// that stamp a surface without a model (legacy) keep their prior meaning.
|
|
83
|
+
function surfaceLabel() { return anthropicSurface(); }
|
|
84
|
+
|
|
33
85
|
function providerName() {
|
|
34
86
|
return (process.env.CLAUDE_PROVIDER || 'cli').toLowerCase();
|
|
35
87
|
}
|
|
36
88
|
|
|
37
|
-
//
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
}
|
|
41
|
-
|
|
42
|
-
//
|
|
43
|
-
//
|
|
44
|
-
//
|
|
45
|
-
//
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
89
|
+
// A surface whose spend is metered (real dollars) vs a subscription surface
|
|
90
|
+
// (metered spend $0; the $ figure is the estimated-equivalent API cost).
|
|
91
|
+
function isMeteredSurface(surface) { return surface === 'api' || surface === 'openai-api'; }
|
|
92
|
+
function isSubscriptionSurface(surface) { return !isMeteredSurface(surface); }
|
|
93
|
+
|
|
94
|
+
// Per-surface retry/timeout policy. A cli/subscription surface spawns a first-party
|
|
95
|
+
// CLI subprocess (cold-start-dominated, occasionally hangs), so it gets a longer
|
|
96
|
+
// per-call timeout (300s) and a longer exponential backoff (baseDelay 30s → 30/60/
|
|
97
|
+
// 120s across the 4 retries); api surfaces keep the tighter 120s / 3s defaults.
|
|
98
|
+
function retryPolicyForSurface(surface) {
|
|
99
|
+
const cli = isSubscriptionSurface(surface);
|
|
100
|
+
return { timeoutMs: cli ? 300000 : 120000, baseDelayMs: cli ? 30000 : 3000, tries: 4 };
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
// The fixed harness preamble Codex prepends to every `codex exec` call. Recorded
|
|
104
|
+
// into openai/cli receipts as run.surface_overhead_note so a reader knows the
|
|
105
|
+
// per-call input-token count is dominated by a constant we do not control.
|
|
106
|
+
const CODEX_OVERHEAD_NOTE =
|
|
107
|
+
'openai/cli surface (codex exec): every call carries a fixed Codex base-instruction '
|
|
108
|
+
+ 'preamble of ~12,000–15,000 input tokens that the harness prepends and we do not '
|
|
109
|
+
+ 'control; the model id is set by us via -m (it is not echoed in the JSONL stream). '
|
|
110
|
+
+ 'Approval prompting is off by default on `codex exec` (no -a flag is passed).';
|
|
111
|
+
|
|
112
|
+
// Send a single-turn prompt and return { text, usage, surface, provider }.
|
|
113
|
+
// usage is { input_tokens, output_tokens } when the surface reports it (api
|
|
114
|
+
// surfaces), else null (cli surfaces do not expose token counts to us).
|
|
115
|
+
// `temperature` is honoured ONLY on api surfaces; cli surfaces control sampling
|
|
116
|
+
// themselves, so temperature is ignored there and the receipt records that fact.
|
|
117
|
+
async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = null, temperature = undefined }) {
|
|
118
|
+
const surface = surfaceForModel(model);
|
|
119
|
+
const provider = providerForSurface(surface);
|
|
120
|
+
// Offline stub surface: canned completion, zero model calls. The receipt still
|
|
121
|
+
// records the real surface/provider so a stub run is not mistaken for a genuine
|
|
122
|
+
// one at read time — only the generation/judge TEXT is canned.
|
|
123
|
+
if (stubEnabled()) return { ...stubComplete({ system, prompt }), surface, provider, attempts: 1 };
|
|
124
|
+
|
|
125
|
+
// Per-surface tolerance (see retryPolicyForSurface): a cli/subscription surface
|
|
126
|
+
// spawns a first-party CLI subprocess whose cold-start is slow and occasionally
|
|
127
|
+
// hangs, so it gets a longer per-call timeout (300s) and a longer exponential
|
|
128
|
+
// backoff between the 4 retries (30s / 60s / 120s); api surfaces keep the tighter
|
|
129
|
+
// defaults. An explicit caller timeoutMs still wins.
|
|
130
|
+
const policy = retryPolicyForSurface(surface);
|
|
131
|
+
const effTimeout = timeoutMs != null ? timeoutMs : policy.timeoutMs;
|
|
132
|
+
const baseDelayMs = policy.baseDelayMs;
|
|
133
|
+
|
|
134
|
+
const laneRunner = () => {
|
|
135
|
+
switch (surface) {
|
|
136
|
+
case 'api': return completeAnthropicApi({ system, prompt, model, maxTokens, temperature });
|
|
137
|
+
case 'claude-cli': return completeClaudeCli({ system, prompt, model, timeoutMs: effTimeout });
|
|
138
|
+
case 'openai-api': return completeOpenaiApi({ system, prompt, model, maxTokens, temperature, timeoutMs: effTimeout });
|
|
139
|
+
case 'openai-cli': return completeCodexCli({ system, prompt, model, timeoutMs: effTimeout });
|
|
140
|
+
default: throw new Error(`unknown surface: ${surface}`);
|
|
141
|
+
}
|
|
142
|
+
};
|
|
57
143
|
// A published run makes ~1000s of calls; be patient with transient
|
|
58
|
-
// throttling/cold-starts so one blip doesn't abort a multi-hour grind.
|
|
59
|
-
|
|
60
|
-
|
|
144
|
+
// throttling/cold-starts so one blip doesn't abort a multi-hour grind. `attempts`
|
|
145
|
+
// counts every try (retries included) so the budget can charge for them.
|
|
146
|
+
let attempts = 0;
|
|
147
|
+
const runner = () => { attempts += 1; return withTimeout(laneRunner, effTimeout, `provider(${surface})`); };
|
|
148
|
+
try {
|
|
149
|
+
const out = await withRetry(runner, { tries: policy.tries, baseDelayMs });
|
|
150
|
+
return { ...out, surface, provider, attempts };
|
|
151
|
+
} catch (e) {
|
|
152
|
+
// Surface the attempt count so a persistently-failing call can be charged for
|
|
153
|
+
// (and, for a timeout, marked failed_timeout by the runner instead of fatal).
|
|
154
|
+
if (e && typeof e === 'object') e.attempts = attempts;
|
|
155
|
+
throw e;
|
|
156
|
+
}
|
|
61
157
|
}
|
|
62
158
|
|
|
63
|
-
|
|
159
|
+
// ── anthropic/api ─────────────────────────────────────────────────────────────
|
|
160
|
+
async function completeAnthropicApi({ system, prompt, model, maxTokens, temperature }) {
|
|
64
161
|
let Anthropic;
|
|
65
162
|
try { Anthropic = require('@anthropic-ai/sdk'); }
|
|
66
|
-
catch (_e) { throw new Error('
|
|
67
|
-
if (!process.env.ANTHROPIC_API_KEY) throw new Error('
|
|
163
|
+
catch (_e) { throw new Error('the anthropic/api surface requires the @anthropic-ai/sdk package (npm i @anthropic-ai/sdk)'); }
|
|
164
|
+
if (!process.env.ANTHROPIC_API_KEY) throw new Error('the anthropic/api surface requires ANTHROPIC_API_KEY');
|
|
68
165
|
const client = new Anthropic({ apiKey: process.env.ANTHROPIC_API_KEY });
|
|
69
166
|
const params = {
|
|
70
167
|
model: resolveModel(model),
|
|
@@ -84,7 +181,8 @@ async function completeApi({ system, prompt, model, maxTokens, temperature }) {
|
|
|
84
181
|
};
|
|
85
182
|
}
|
|
86
183
|
|
|
87
|
-
|
|
184
|
+
// ── anthropic/cli (claude -p) ──────────────────────────────────────────────────
|
|
185
|
+
function completeClaudeCli({ system, prompt, model, timeoutMs }) {
|
|
88
186
|
return new Promise((resolve, reject) => {
|
|
89
187
|
// Strip ANTHROPIC_API_KEY so the CLI uses the subscription session, not the
|
|
90
188
|
// metered API key. Everything else in the env is preserved.
|
|
@@ -102,7 +200,7 @@ function completeCli({ system, prompt, model, timeoutMs }) {
|
|
|
102
200
|
child.stderr.on('data', (d) => { err += d; });
|
|
103
201
|
child.on('error', (e) => {
|
|
104
202
|
clearTimeout(killer);
|
|
105
|
-
if (e.code === 'ENOENT') reject(new Error("
|
|
203
|
+
if (e.code === 'ENOENT') reject(new Error("the anthropic/cli surface requires the `claude` CLI on PATH. Install it or set CLAUDE_PROVIDER=api."));
|
|
106
204
|
else reject(e);
|
|
107
205
|
});
|
|
108
206
|
child.on('close', (code) => {
|
|
@@ -115,4 +213,149 @@ function completeCli({ system, prompt, model, timeoutMs }) {
|
|
|
115
213
|
});
|
|
116
214
|
}
|
|
117
215
|
|
|
118
|
-
|
|
216
|
+
// ── openai/api (Chat Completions-compatible) ───────────────────────────────────
|
|
217
|
+
// Generic OpenAI-compatible client using global fetch (Node ≥ 18). The base_url
|
|
218
|
+
// is configurable — env OPENAI_BASE_URL wins, else the registry's provider
|
|
219
|
+
// base_url (lib/models.providerConfig), else api.openai.com/v1. Building it
|
|
220
|
+
// base_url-first is deliberate: a future provider is a registry entry with a
|
|
221
|
+
// base_url + api-key env, not a new code path.
|
|
222
|
+
async function completeOpenaiApi({ system, prompt, model, maxTokens, temperature, timeoutMs }) {
|
|
223
|
+
const cfg = openaiProviderConfig();
|
|
224
|
+
const apiKey = process.env[cfg.apiKeyEnv] || process.env.OPENAI_API_KEY;
|
|
225
|
+
if (!apiKey) throw new Error(`the openai/api surface requires ${cfg.apiKeyEnv} (or run the codex CLI surface). Set OPENAI_SURFACE=cli to use the subscription instead.`);
|
|
226
|
+
const baseUrl = (process.env.OPENAI_BASE_URL || cfg.baseUrl || 'https://api.openai.com/v1').replace(/\/+$/, '');
|
|
227
|
+
const messages = [];
|
|
228
|
+
if (system) messages.push({ role: 'system', content: system });
|
|
229
|
+
messages.push({ role: 'user', content: prompt });
|
|
230
|
+
const body = { model: resolveModel(model), messages };
|
|
231
|
+
// Newer OpenAI models use `max_completion_tokens`; classic Chat Completions use
|
|
232
|
+
// `max_tokens`. Send the modern field; harmless on classic-compatible servers
|
|
233
|
+
// that ignore unknown fields, and correct for current OpenAI models.
|
|
234
|
+
if (maxTokens) body.max_completion_tokens = maxTokens;
|
|
235
|
+
if (temperature !== undefined) body.temperature = temperature;
|
|
236
|
+
|
|
237
|
+
const controller = new AbortController();
|
|
238
|
+
const t = setTimeout(() => controller.abort(), timeoutMs);
|
|
239
|
+
let resp;
|
|
240
|
+
try {
|
|
241
|
+
resp = await fetch(`${baseUrl}/chat/completions`, {
|
|
242
|
+
method: 'POST',
|
|
243
|
+
headers: { 'content-type': 'application/json', authorization: `Bearer ${apiKey}` },
|
|
244
|
+
body: JSON.stringify(body),
|
|
245
|
+
signal: controller.signal,
|
|
246
|
+
});
|
|
247
|
+
} finally { clearTimeout(t); }
|
|
248
|
+
const raw = await resp.text();
|
|
249
|
+
if (!resp.ok) throw new Error(`openai/api ${resp.status}: ${raw.slice(0, 300)}`);
|
|
250
|
+
let parsed;
|
|
251
|
+
try { parsed = JSON.parse(raw); } catch (_e) { throw new Error(`openai/api: unparseable response: ${raw.slice(0, 200)}`); }
|
|
252
|
+
const text = ((parsed.choices && parsed.choices[0] && parsed.choices[0].message && parsed.choices[0].message.content) || '');
|
|
253
|
+
const usage = parsed.usage || {};
|
|
254
|
+
return {
|
|
255
|
+
text: String(text).trim(),
|
|
256
|
+
usage: {
|
|
257
|
+
input_tokens: usage.prompt_tokens || 0,
|
|
258
|
+
output_tokens: usage.completion_tokens || 0,
|
|
259
|
+
},
|
|
260
|
+
};
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
// Read the OpenAI provider config (base_url + api-key env) from the registry,
|
|
264
|
+
// lazily (require inside the fn) so provider.js keeps no load-time cycle with
|
|
265
|
+
// lib/models.js. Falls back to the public defaults if the registry omits it.
|
|
266
|
+
function openaiProviderConfig() {
|
|
267
|
+
try {
|
|
268
|
+
const cfg = require('./models').providerConfig('openai');
|
|
269
|
+
if (cfg) return { baseUrl: cfg.base_url, apiKeyEnv: cfg.api_key_env || 'OPENAI_API_KEY' };
|
|
270
|
+
} catch (_e) { /* fall through to defaults */ }
|
|
271
|
+
return { baseUrl: 'https://api.openai.com/v1', apiKeyEnv: 'OPENAI_API_KEY' };
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
// ── openai/cli (codex exec — ChatGPT subscription) ─────────────────────────────
|
|
275
|
+
// Invocation template (from the verified recon, reference_codex_cli):
|
|
276
|
+
// codex exec --json -s read-only --skip-git-repo-check --ephemeral -o <tmp> [-m <model>] "PROMPT"
|
|
277
|
+
// Honoured facts:
|
|
278
|
+
// - `-a/--ask-for-approval` is INVALID on `codex exec` — it is NEVER passed;
|
|
279
|
+
// exec already defaults to approval_policy "never".
|
|
280
|
+
// - the model id is NOT echoed in the JSONL stream; we set -m so the receipt
|
|
281
|
+
// records what WE requested (run.model_id).
|
|
282
|
+
// - auth lives at ~/.codex/auth.json; we do a PRESENCE check only and NEVER
|
|
283
|
+
// read or print its contents.
|
|
284
|
+
// - `codex exec` has no system-prompt flag, so a system prompt (the SKILL.md in
|
|
285
|
+
// with_skill mode) is folded into the prompt text.
|
|
286
|
+
// - stderr emits a harmless "Reading additional input from stdin…" notice.
|
|
287
|
+
const CODEX_EXEC_ARGS = ['exec', '--json', '-s', 'read-only', '--skip-git-repo-check', '--ephemeral'];
|
|
288
|
+
|
|
289
|
+
function codexAuthPresent(homeDir) {
|
|
290
|
+
const home = homeDir || process.env.CODEX_HOME_DIR || os.homedir();
|
|
291
|
+
// CODEX_HOME overrides the auth location if the user set it (codex convention).
|
|
292
|
+
const base = process.env.CODEX_HOME || path.join(home, '.codex');
|
|
293
|
+
return fs.existsSync(path.join(base, 'auth.json'));
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
// Build the exact argv for a codex exec call. Exposed for the gate to assert the
|
|
297
|
+
// template (and that `-a` never appears). `outFile` receives the final message.
|
|
298
|
+
function buildCodexArgs({ model, outFile }) {
|
|
299
|
+
const args = [...CODEX_EXEC_ARGS, '-o', outFile];
|
|
300
|
+
if (model) args.push('-m', resolveModel(model));
|
|
301
|
+
return args;
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
// The FULL argv for a codex call: the template plus a single `-` positional that
|
|
305
|
+
// means "read the prompt from stdin". The prompt is NEVER a positional argv —
|
|
306
|
+
// real SKILL.md files start with `---` (YAML frontmatter) and codex rejects a
|
|
307
|
+
// positional beginning with `--`. Exposed so the gate can lock this in.
|
|
308
|
+
function codexFinalArgs({ model, outFile }) {
|
|
309
|
+
return [...buildCodexArgs({ model, outFile }), '-'];
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
function completeCodexCli({ system, prompt, model, timeoutMs }) {
|
|
313
|
+
return new Promise((resolve, reject) => {
|
|
314
|
+
if (!codexAuthPresent()) {
|
|
315
|
+
return reject(new Error('the openai/cli surface requires codex auth (~/.codex/auth.json). Run `codex login` (or `codex login --device-auth` on a headless box).'));
|
|
316
|
+
}
|
|
317
|
+
// codex exec has no system-prompt flag → fold the system prompt into the
|
|
318
|
+
// prompt text (this is how the SKILL.md reaches the model in with_skill mode).
|
|
319
|
+
const fullPrompt = system ? `${system}\n\n---\n\n${prompt}` : prompt;
|
|
320
|
+
const outFile = path.join(os.tmpdir(), `driftproof-codex-${process.pid}-${Date.now()}-${Math.floor(codexCounter())}.txt`);
|
|
321
|
+
// The prompt is delivered on STDIN, not as a positional argv: real SKILL.md
|
|
322
|
+
// files begin with `---` (YAML frontmatter), and a positional argument that
|
|
323
|
+
// starts with `--` is rejected by codex's arg parser ("unexpected argument
|
|
324
|
+
// '---'"). `-` as the positional tells `codex exec` to read instructions from
|
|
325
|
+
// stdin (per its --help), which is content-agnostic.
|
|
326
|
+
const args = codexFinalArgs({ model, outFile });
|
|
327
|
+
|
|
328
|
+
const child = spawn('codex', args, { env: { ...process.env }, stdio: ['pipe', 'ignore', 'pipe'] });
|
|
329
|
+
let err = '';
|
|
330
|
+
const killer = setTimeout(() => { child.kill('SIGKILL'); }, timeoutMs + 5000);
|
|
331
|
+
child.stderr.on('data', (d) => { err += d; });
|
|
332
|
+
child.on('error', (e) => {
|
|
333
|
+
clearTimeout(killer);
|
|
334
|
+
if (e.code === 'ENOENT') reject(new Error('the openai/cli surface requires the `codex` CLI on PATH (npm i -g @openai/codex).'));
|
|
335
|
+
else reject(e);
|
|
336
|
+
});
|
|
337
|
+
child.on('close', (code) => {
|
|
338
|
+
clearTimeout(killer);
|
|
339
|
+
let text = '';
|
|
340
|
+
try { text = fs.readFileSync(outFile, 'utf8'); } catch (_e) { /* fall through */ }
|
|
341
|
+
try { fs.unlinkSync(outFile); } catch (_e) { /* best effort */ }
|
|
342
|
+
if (code !== 0) return reject(new Error(`codex exec exited ${code}: ${String(err).slice(0, 400)}`));
|
|
343
|
+
resolve({ text: String(text).trim(), usage: null });
|
|
344
|
+
});
|
|
345
|
+
child.stdin.on('error', () => { /* ignore EPIPE if codex exits before reading all input */ });
|
|
346
|
+
child.stdin.write(fullPrompt);
|
|
347
|
+
child.stdin.end();
|
|
348
|
+
});
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
// A monotonically-increasing counter for temp-file uniqueness that does not use
|
|
352
|
+
// Math.random (kept deterministic-friendly for any harness that forbids it).
|
|
353
|
+
let _codexCounter = 0;
|
|
354
|
+
function codexCounter() { _codexCounter += 1; return _codexCounter; }
|
|
355
|
+
|
|
356
|
+
module.exports = {
|
|
357
|
+
complete, resolveModel, surfaceLabel, providerName, MODEL_ALIASES,
|
|
358
|
+
inferProvider, surfaceForModel, providerForSurface,
|
|
359
|
+
isMeteredSurface, isSubscriptionSurface, retryPolicyForSurface,
|
|
360
|
+
CODEX_OVERHEAD_NOTE, CODEX_EXEC_ARGS, buildCodexArgs, codexFinalArgs, codexAuthPresent,
|
|
361
|
+
};
|
package/lib/receipt.js
CHANGED
|
@@ -13,7 +13,8 @@ const { aggregateBands, combineUncertainty, round } = require('./stats');
|
|
|
13
13
|
const SCHEMA_FILES = {
|
|
14
14
|
'0.1': 'receipt.v0.1.schema.json',
|
|
15
15
|
'0.2': 'receipt.v0.2.schema.json',
|
|
16
|
-
'0.3': 'receipt.schema.json',
|
|
16
|
+
'0.3': 'receipt.v0.3.schema.json',
|
|
17
|
+
'0.3.1': 'receipt.schema.json',
|
|
17
18
|
};
|
|
18
19
|
|
|
19
20
|
const _validators = {};
|
|
@@ -80,8 +81,12 @@ function aggregate(caseResults) {
|
|
|
80
81
|
// cases: [ { id, mode, outcome, score, mean, stddev, samples, generation_hash, judge_sample_hashes, threshold, reason, judge } ]
|
|
81
82
|
// editorialReviews: optional [ { url, source, date } ]
|
|
82
83
|
function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED', editorialReviews = null }) {
|
|
83
|
-
|
|
84
|
-
|
|
84
|
+
// v0.3.1: cases marked failed_timeout are recorded in results.cases but EXCLUDED
|
|
85
|
+
// from aggregates — a band is never fabricated from a case that did not complete.
|
|
86
|
+
const okCases = cases.filter((c) => c.case_status !== 'failed_timeout');
|
|
87
|
+
const failedCount = cases.length - okCases.length;
|
|
88
|
+
const withSkill = okCases.filter((c) => c.mode === 'with_skill');
|
|
89
|
+
const baseline = okCases.filter((c) => c.mode === 'baseline');
|
|
85
90
|
const aggWith = aggregate(withSkill);
|
|
86
91
|
const aggBase = aggregate(baseline);
|
|
87
92
|
|
|
@@ -100,6 +105,9 @@ function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED',
|
|
|
100
105
|
run: {
|
|
101
106
|
model_id: run.model_id,
|
|
102
107
|
model_release_date: run.model_release_date == null ? null : run.model_release_date,
|
|
108
|
+
// v0.3.1: two-axis provider (registry `provider`, else inferred). Defaults
|
|
109
|
+
// to anthropic for any legacy caller that omits it.
|
|
110
|
+
provider: run.provider || 'anthropic',
|
|
103
111
|
surface: run.surface,
|
|
104
112
|
runner_version: run.runner_version,
|
|
105
113
|
date_utc: run.date_utc,
|
|
@@ -124,6 +132,17 @@ function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED',
|
|
|
124
132
|
verification_level: verificationLevel,
|
|
125
133
|
receipt_hash: '',
|
|
126
134
|
};
|
|
135
|
+
// v0.3.1 additive-optional fields (canonicalization sorts keys, so placement
|
|
136
|
+
// here does not affect the hash):
|
|
137
|
+
// run.surface_overhead_note — the fixed harness preamble on the openai/cli surface.
|
|
138
|
+
// skill.tokens — estimated SKILL.md token size (value-per-token axis).
|
|
139
|
+
if (run.surface_overhead_note) receipt.run.surface_overhead_note = run.surface_overhead_note;
|
|
140
|
+
if (skill.tokens != null) receipt.skill.tokens = skill.tokens;
|
|
141
|
+
// v0.3.1: mark the receipt incomplete when any case failed (excluded above).
|
|
142
|
+
if (failedCount > 0) {
|
|
143
|
+
receipt.run.status = 'incomplete';
|
|
144
|
+
receipt.run.failed_case_count = failedCount;
|
|
145
|
+
}
|
|
127
146
|
if (editorialReviews && editorialReviews.length) receipt.editorial_reviews = editorialReviews;
|
|
128
147
|
return sealReceipt(receipt);
|
|
129
148
|
}
|
package/lib/run.js
CHANGED
|
@@ -1,12 +1,14 @@
|
|
|
1
1
|
// SPDX-License-Identifier: Apache-2.0
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
-
const { complete, resolveModel,
|
|
4
|
+
const { complete, resolveModel, surfaceForModel, CODEX_OVERHEAD_NOTE } = require('./provider');
|
|
5
5
|
const { gradeSamples, judgeSettings } = require('./judge');
|
|
6
6
|
const { buildReceipt } = require('./receipt');
|
|
7
7
|
const { sha256 } = require('./canonical');
|
|
8
|
-
const { registryStatus } = require('./models');
|
|
8
|
+
const { registryStatus, providerForModel } = require('./models');
|
|
9
9
|
const { perCallCostUSD } = require('./cost');
|
|
10
|
+
const { runChecks } = require('./checks');
|
|
11
|
+
const { estimateTokens } = require('./skillCost');
|
|
10
12
|
const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES } = require('../config');
|
|
11
13
|
|
|
12
14
|
// Known model release dates (best-effort; null when unknown). Recorded into the
|
|
@@ -20,6 +22,8 @@ const MODEL_RELEASE_DATES = {
|
|
|
20
22
|
'claude-haiku-4-5': '2025-10-01',
|
|
21
23
|
'claude-sonnet-5': '2026-06-30', // anthropic.com/news/claude-sonnet-5
|
|
22
24
|
'claude-sonnet-4-6': '2026-02-17', // anthropic.com/news/claude-sonnet-4-6
|
|
25
|
+
'claude-opus-5': '2026-07-24', // anthropic.com/news/claude-opus-5
|
|
26
|
+
'claude-opus-4-8': '2026-05-28', // anthropic.com/news/claude-opus-4-8
|
|
23
27
|
};
|
|
24
28
|
|
|
25
29
|
function releaseDateFor(modelId) {
|
|
@@ -39,9 +43,16 @@ function projectCalls(caseCount, samples) {
|
|
|
39
43
|
// SKILL.md is prepended as a system prompt (the whole point: measure the skill's
|
|
40
44
|
// marginal effect vs a bare baseline).
|
|
41
45
|
async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs }) {
|
|
46
|
+
// Test seam (gate only): force a persistent timeout for a named case id so the
|
|
47
|
+
// failed_timeout path is exercised deterministically without any live call.
|
|
48
|
+
if (process.env.DRIFTPROOF_TEST_TIMEOUT_CASEID && process.env.DRIFTPROOF_TEST_TIMEOUT_CASEID === caseObj.id) {
|
|
49
|
+
const e = new Error('provider timed out (test seam) after retries');
|
|
50
|
+
e.code = 'TIMEOUT'; e.attempts = 4;
|
|
51
|
+
throw e;
|
|
52
|
+
}
|
|
42
53
|
const system = withSkill ? skillMd : undefined;
|
|
43
|
-
const { text, usage } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
|
|
44
|
-
return { text, usage };
|
|
54
|
+
const { text, usage, attempts } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
|
|
55
|
+
return { text, usage, attempts };
|
|
45
56
|
}
|
|
46
57
|
|
|
47
58
|
// Determine a case outcome from its sampled band and threshold.
|
|
@@ -77,7 +88,10 @@ async function judgeCase({ caseObj, response, generationHash, judgeModel, mode,
|
|
|
77
88
|
reason: g.reason,
|
|
78
89
|
judge: { model_id: g.model_id, rubric_hash: g.rubric_hash },
|
|
79
90
|
};
|
|
80
|
-
|
|
91
|
+
// v0.3.1 deterministic post-checks (supplementary; NOT folded into `outcome`).
|
|
92
|
+
const checks = runChecks(response, caseObj.checks);
|
|
93
|
+
if (checks.length) caseResult.checks = checks;
|
|
94
|
+
return { caseResult, sampleTexts: g.sample_texts, attempts: g.attempts };
|
|
81
95
|
}
|
|
82
96
|
|
|
83
97
|
// Run up to `concurrency` async tasks at a time, preserving input order in the
|
|
@@ -112,6 +126,9 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
112
126
|
const modelId = resolveModel(model);
|
|
113
127
|
const judgeModel = opts.judgeModel ? resolveModel(opts.judgeModel) : modelId;
|
|
114
128
|
const timeoutMs = opts.timeoutMs || 120000;
|
|
129
|
+
// Optional per-case timeout overrides { caseId: ms }; a slow case can get a
|
|
130
|
+
// longer budget without lengthening every other case's per-call timeout.
|
|
131
|
+
const caseTimeoutMs = opts.caseTimeoutMs || {};
|
|
115
132
|
const maxCalls = opts.maxCalls || 200;
|
|
116
133
|
const samples = opts.samples || DEFAULT_JUDGE_SAMPLES;
|
|
117
134
|
const concurrency = Math.max(1, opts.concurrency || 1);
|
|
@@ -137,46 +154,75 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
137
154
|
for (const c of cases) for (const withSkill of [true, false]) tasks.push({ c, withSkill });
|
|
138
155
|
|
|
139
156
|
let calls = 0;
|
|
157
|
+
let failedCases = 0;
|
|
158
|
+
const isTimeout = (e) => !!(e && (e.code === 'TIMEOUT' || /tim(e|ed)\s*out/i.test(String((e && e.message) || ''))));
|
|
159
|
+
const genKind = (ws) => (ws ? 'gen_with_skill' : 'gen_baseline');
|
|
140
160
|
const pairs = await mapPool(tasks, concurrency, async ({ c, withSkill }) => {
|
|
141
161
|
const mode = withSkill ? 'with_skill' : 'baseline';
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
:
|
|
157
|
-
|
|
162
|
+
const ct = caseTimeoutMs[c.id] || timeoutMs;
|
|
163
|
+
try {
|
|
164
|
+
onProgress({ case: c.id, mode, phase: 'generate' });
|
|
165
|
+
const gen = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs: ct });
|
|
166
|
+
calls += 1;
|
|
167
|
+
// Live budget: count the generation call INCLUDING retries, then hard-stop
|
|
168
|
+
// if over 1.25× cap.
|
|
169
|
+
if (budget) budget.add((gen.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
|
|
170
|
+
const generationHash = sha256(String(gen.text || ''));
|
|
171
|
+
onProgress({ case: c.id, mode, phase: 'judge', samples });
|
|
172
|
+
const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ct, samples });
|
|
173
|
+
calls += samples;
|
|
174
|
+
// Live budget: count all judge calls (retries included) for this (case, mode).
|
|
175
|
+
if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judgeModel, 'judge'));
|
|
176
|
+
onProgress({ case: c.id, mode, phase: 'done', outcome: jr.caseResult.outcome, score: jr.caseResult.mean, stddev: jr.caseResult.stddev });
|
|
177
|
+
const transcript = keepTranscripts
|
|
178
|
+
? { id: c.id, mode, generation: String(gen.text || ''), judge_outputs: jr.sampleTexts }
|
|
179
|
+
: null;
|
|
180
|
+
return { caseResult: jr.caseResult, transcript };
|
|
181
|
+
} catch (e) {
|
|
182
|
+
if (e && e.code === 'BUDGET_HARDSTOP') throw e; // budget hard-stop stays fatal
|
|
183
|
+
if (!isTimeout(e)) throw e; // non-timeout errors stay fatal
|
|
184
|
+
// Persistent timeout → NON-FATAL: charge the consumed attempts, record the
|
|
185
|
+
// case as failed_timeout (no fabricated samples), and continue the run.
|
|
186
|
+
if (budget) {
|
|
187
|
+
try {
|
|
188
|
+
if (e.phase === 'judge') budget.add((e.judgeAttempts || 1) * perCallCostUSD(judgeModel, 'judge'));
|
|
189
|
+
else budget.add((e.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
|
|
190
|
+
} catch (be) { if (be && be.code === 'BUDGET_HARDSTOP') throw be; }
|
|
191
|
+
}
|
|
192
|
+
failedCases += 1;
|
|
193
|
+
onProgress({ case: c.id, mode, phase: 'failed', reason: String((e && e.message) || 'timeout') });
|
|
194
|
+
return { caseResult: { id: c.id, mode, case_status: 'failed_timeout', reason: String((e && e.message) || 'timeout').slice(0, 200) }, transcript: null };
|
|
195
|
+
}
|
|
158
196
|
});
|
|
159
197
|
const caseResults = pairs.map((p) => p.caseResult);
|
|
160
198
|
const transcripts = keepTranscripts ? pairs.map((p) => p.transcript) : null;
|
|
161
199
|
|
|
200
|
+
const surface = surfaceForModel(modelId);
|
|
162
201
|
const receipt = buildReceipt({
|
|
163
|
-
skill: {
|
|
202
|
+
skill: {
|
|
203
|
+
name: skill.name, version: skill.version, contentHash: skill.contentHash,
|
|
204
|
+
// v0.3.1 value-per-token axis: estimated SKILL.md token size.
|
|
205
|
+
tokens: estimateTokens(skill.skillMd),
|
|
206
|
+
},
|
|
164
207
|
suite: { format: skill.suite.format, suiteHash: skill.suite.suiteHash, caseCount: skill.suite.caseCount },
|
|
165
208
|
run: {
|
|
166
209
|
model_id: modelId,
|
|
167
210
|
model_release_date: releaseDateFor(modelId),
|
|
168
|
-
|
|
211
|
+
provider: providerForModel(modelId),
|
|
212
|
+
surface,
|
|
213
|
+
// v0.3.1: on the openai/cli (codex) surface, record the fixed harness preamble.
|
|
214
|
+
surface_overhead_note: surface === 'openai-cli' ? CODEX_OVERHEAD_NOTE : undefined,
|
|
169
215
|
runner_version: RUNNER_VERSION,
|
|
170
216
|
date_utc: opts.nowIso || new Date().toISOString(),
|
|
171
217
|
registry: registryStatus(modelId),
|
|
172
218
|
transcripts: keepTranscripts ? 'retained-local' : 'hashes-only',
|
|
173
|
-
judge: judgeSettings(samples),
|
|
219
|
+
judge: judgeSettings(samples, judgeModel),
|
|
174
220
|
},
|
|
175
221
|
cases: caseResults,
|
|
176
222
|
verificationLevel: 'TESTED',
|
|
177
223
|
});
|
|
178
224
|
|
|
179
|
-
return { receipt, calls, transcripts };
|
|
225
|
+
return { receipt, calls, transcripts, failedCases };
|
|
180
226
|
}
|
|
181
227
|
|
|
182
228
|
function band(mean, sd) { return `${mean.toFixed(3)} ± ${sd.toFixed(3)}`; }
|
|
@@ -191,7 +237,7 @@ function summarizeReceipt(receipt) {
|
|
|
191
237
|
L.push(`- **run (UTC):** ${receipt.run.date_utc}`);
|
|
192
238
|
L.push(`- **runner:** v${receipt.run.runner_version}`);
|
|
193
239
|
const j = receipt.run.judge || {};
|
|
194
|
-
L.push(`- **judge:** ${j.samples || 1} samples/case, temperature ${j.temperature == null ? 'n/a
|
|
240
|
+
L.push(`- **judge:** ${j.samples || 1} samples/case, temperature ${j.temperature == null ? 'n/a' : j.temperature} (${j.sampling || 'single'})`);
|
|
195
241
|
if (receipt.run.registry) L.push(`- **registry:** ${receipt.run.registry} **transcripts:** ${receipt.run.transcripts || 'hashes-only'}`);
|
|
196
242
|
L.push(`- **skill content_hash:** \`${receipt.skill.content_hash.slice(0, 16)}…\``);
|
|
197
243
|
L.push(`- **suite:** ${receipt.suite.case_count} cases (${receipt.suite.format})`);
|
|
@@ -203,6 +249,10 @@ function summarizeReceipt(receipt) {
|
|
|
203
249
|
const cmp = receipt.comparison;
|
|
204
250
|
const aggs = receipt.results.aggregates;
|
|
205
251
|
const sign = cmp.delta >= 0 ? '+' : '';
|
|
252
|
+
if (receipt.run.status === 'incomplete') {
|
|
253
|
+
L.push(`> ⏱ **INCOMPLETE** — ${receipt.run.failed_case_count} case(s) failed (timed out after retries) and are EXCLUDED from the aggregates below; this receipt must not be used to compute a drift/durability verdict.`);
|
|
254
|
+
L.push('');
|
|
255
|
+
}
|
|
206
256
|
L.push(`with_skill **${band(aggs.with_skill.mean_score, aggs.with_skill.stddev)}** vs baseline **${band(aggs.baseline.mean_score, aggs.baseline.stddev)}**`);
|
|
207
257
|
L.push('');
|
|
208
258
|
L.push(`skill lift **${sign}${cmp.delta.toFixed(3)}** (combined uncertainty ± ${cmp.delta_uncertainty.toFixed(3)})`);
|
|
@@ -212,6 +262,10 @@ function summarizeReceipt(receipt) {
|
|
|
212
262
|
L.push(`| case | mode | outcome | mean ± stddev | judge reason |`);
|
|
213
263
|
L.push(`|---|---|---|---|---|`);
|
|
214
264
|
for (const c of receipt.results.cases) {
|
|
265
|
+
if (c.case_status === 'failed_timeout') {
|
|
266
|
+
L.push(`| \`${c.id}\` | ${c.mode} | ⏱ failed_timeout | — (not measured) | ${c.reason || 'timed out'} |`);
|
|
267
|
+
continue;
|
|
268
|
+
}
|
|
215
269
|
const flag = c.outcome === 'borderline' ? ' ⚠' : '';
|
|
216
270
|
L.push(`| \`${c.id}\` | ${c.mode} | ${c.outcome}${flag} | ${band(c.mean, c.stddev || 0)} | ${c.reason || ''} |`);
|
|
217
271
|
}
|