driftproof 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/provider.js CHANGED
@@ -1,20 +1,33 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
+ const fs = require('fs');
5
+ const os = require('os');
6
+ const path = require('path');
4
7
  const { spawn } = require('child_process');
5
8
  const { withRetry, withTimeout } = require('./json');
6
9
  const { stubComplete, stubEnabled } = require('./stub');
7
10
 
8
- // Provider abstraction: one `complete()` call, two surfaces.
11
+ // Provider abstraction: one `complete()` call over a TWO-AXIS lane model —
12
+ // provider (anthropic | openai) × surface (api | cli). Four concrete lanes:
9
13
  //
10
- // CLAUDE_PROVIDER=api → Anthropic Messages API (needs ANTHROPIC_API_KEY).
11
- // CLAUDE_PROVIDER=cli → spawn `claude -p` with ANTHROPIC_API_KEY STRIPPED
12
- // from the child env, so dev runs draw on the local
13
- // subscription session credit rather than metered API
14
- // billing. This is the DEFAULT for dev.
14
+ // anthropic/api → Anthropic Messages API (ANTHROPIC_API_KEY) label "api"
15
+ // anthropic/cli → spawn `claude -p` on the subscription label "claude-cli"
16
+ // openai/api → Chat Completions-compatible API (OPENAI_API_KEY, base_url
17
+ // configurable — generic so a future provider is just a
18
+ // registry entry with a base_url) label "openai-api"
19
+ // openai/cli → spawn `codex exec` on the ChatGPT subscription label "openai-cli"
20
+ //
21
+ // The provider of a call is inferred from the MODEL id (so a run can generate on
22
+ // an OpenAI model while the fixed Haiku judge still runs on Anthropic). The
23
+ // surface for each provider is chosen from the environment:
24
+ //
25
+ // anthropic: CLAUDE_PROVIDER = api | cli (default cli — subscription)
26
+ // openai: OPENAI_SURFACE = api | cli (default: api iff OPENAI_API_KEY
27
+ // is present, else cli/codex)
15
28
  //
16
29
  // The surface actually used is returned so the runner can stamp it into the
17
- // receipt (run.surface = "api" | "claude-cli").
30
+ // receipt (run.surface), together with run.provider.
18
31
 
19
32
  // Short model aliases → canonical ids. Kept tiny and explicit; unknown values
20
33
  // are passed through verbatim so a full model id always works.
@@ -24,47 +37,131 @@ const MODEL_ALIASES = {
24
37
  'sonnet-5': 'claude-sonnet-5',
25
38
  'sonnet-4-6': 'claude-sonnet-4-6', // previous Sonnet point release
26
39
  opus: 'claude-opus-4-8',
40
+ // OpenAI convenience aliases (full ids always work too).
41
+ gpt: 'gpt-5.6-sol',
42
+ 'gpt-flagship': 'gpt-5.6-sol',
27
43
  };
28
44
 
29
45
  function resolveModel(m) {
30
46
  return MODEL_ALIASES[m] || m;
31
47
  }
32
48
 
49
+ // Infer the provider from a (resolved) model id. Pure and self-contained (no
50
+ // registry require) so provider.js has no cycle with lib/models.js; the registry
51
+ // `provider` field is authoritative where a model is registered (see
52
+ // lib/models.providerForModel), and registered OpenAI ids match this prefix too.
53
+ function inferProvider(modelId) {
54
+ const id = String(resolveModel(modelId) || '').toLowerCase();
55
+ if (/^(gpt-|o[1-9]|chatgpt|codex|text-|davinci|omni)/.test(id)) return 'openai';
56
+ return 'anthropic';
57
+ }
58
+
59
+ // ── surface selection per provider ───────────────────────────────────────────
60
+ function anthropicSurface() {
61
+ return (process.env.CLAUDE_PROVIDER || 'cli').toLowerCase() === 'api' ? 'api' : 'claude-cli';
62
+ }
63
+
64
+ function openaiSurface() {
65
+ const pref = (process.env.OPENAI_SURFACE || '').toLowerCase();
66
+ if (pref === 'api') return 'openai-api';
67
+ if (pref === 'cli' || pref === 'codex') return 'openai-cli';
68
+ // Auto: prefer the metered API when a key is present (published-run default),
69
+ // else fall back to the Codex subscription surface.
70
+ return process.env.OPENAI_API_KEY ? 'openai-api' : 'openai-cli';
71
+ }
72
+
73
+ function surfaceForModel(modelId) {
74
+ return inferProvider(modelId) === 'openai' ? openaiSurface() : anthropicSurface();
75
+ }
76
+
77
+ function providerForSurface(surface) {
78
+ return (surface === 'openai-api' || surface === 'openai-cli') ? 'openai' : 'anthropic';
79
+ }
80
+
81
+ // Backward-compatible default surface label (the Anthropic axis). Kept so callers
82
+ // that stamp a surface without a model (legacy) keep their prior meaning.
83
+ function surfaceLabel() { return anthropicSurface(); }
84
+
33
85
  function providerName() {
34
86
  return (process.env.CLAUDE_PROVIDER || 'cli').toLowerCase();
35
87
  }
36
88
 
37
- // The receipt-facing label for the surface in use.
38
- function surfaceLabel() {
39
- return providerName() === 'api' ? 'api' : 'claude-cli';
40
- }
41
-
42
- // Send a single-turn prompt and return { text, usage, surface }.
43
- // usage is { input_tokens, output_tokens } when the surface reports it (api),
44
- // else null (cli does not expose token counts to us).
45
- // `temperature` is honoured ONLY on the api surface (the Messages API exposes
46
- // it). On the cli surface sampling params are surface-controlled — we cannot set
47
- // them, so temperature is ignored there and the receipt records that fact.
48
- async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = 120000, temperature = undefined }) {
49
- const surface = surfaceLabel();
50
- // Offline stub surface: return a canned completion with zero model calls. The
51
- // receipt still records the real surface label so a stub run is not mistaken
52
- // for a genuine one at read time — only run.judge/generation TEXT is canned.
53
- if (stubEnabled()) return { ...stubComplete({ system, prompt }), surface };
54
- const runner = () => (surface === 'api'
55
- ? completeApi({ system, prompt, model, maxTokens, temperature })
56
- : completeCli({ system, prompt, model, timeoutMs }));
89
+ // A surface whose spend is metered (real dollars) vs a subscription surface
90
+ // (metered spend $0; the $ figure is the estimated-equivalent API cost).
91
+ function isMeteredSurface(surface) { return surface === 'api' || surface === 'openai-api'; }
92
+ function isSubscriptionSurface(surface) { return !isMeteredSurface(surface); }
93
+
94
+ // Per-surface retry/timeout policy. A cli/subscription surface spawns a first-party
95
+ // CLI subprocess (cold-start-dominated, occasionally hangs), so it gets a longer
96
+ // per-call timeout (300s) and a longer exponential backoff (baseDelay 30s → 30/60/
97
+ // 120s across the 4 retries); api surfaces keep the tighter 120s / 3s defaults.
98
+ function retryPolicyForSurface(surface) {
99
+ const cli = isSubscriptionSurface(surface);
100
+ return { timeoutMs: cli ? 300000 : 120000, baseDelayMs: cli ? 30000 : 3000, tries: 4 };
101
+ }
102
+
103
+ // The fixed harness preamble Codex prepends to every `codex exec` call. Recorded
104
+ // into openai/cli receipts as run.surface_overhead_note so a reader knows the
105
+ // per-call input-token count is dominated by a constant we do not control.
106
+ const CODEX_OVERHEAD_NOTE =
107
+ 'openai/cli surface (codex exec): every call carries a fixed Codex base-instruction '
108
+ + 'preamble of ~12,000–15,000 input tokens that the harness prepends and we do not '
109
+ + 'control; the model id is set by us via -m (it is not echoed in the JSONL stream). '
110
+ + 'Approval prompting is off by default on `codex exec` (no -a flag is passed).';
111
+
112
+ // Send a single-turn prompt and return { text, usage, surface, provider }.
113
+ // usage is { input_tokens, output_tokens } when the surface reports it (api
114
+ // surfaces), else null (cli surfaces do not expose token counts to us).
115
+ // `temperature` is honoured ONLY on api surfaces; cli surfaces control sampling
116
+ // themselves, so temperature is ignored there and the receipt records that fact.
117
+ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = null, temperature = undefined }) {
118
+ const surface = surfaceForModel(model);
119
+ const provider = providerForSurface(surface);
120
+ // Offline stub surface: canned completion, zero model calls. The receipt still
121
+ // records the real surface/provider so a stub run is not mistaken for a genuine
122
+ // one at read time — only the generation/judge TEXT is canned.
123
+ if (stubEnabled()) return { ...stubComplete({ system, prompt }), surface, provider, attempts: 1 };
124
+
125
+ // Per-surface tolerance (see retryPolicyForSurface): a cli/subscription surface
126
+ // spawns a first-party CLI subprocess whose cold-start is slow and occasionally
127
+ // hangs, so it gets a longer per-call timeout (300s) and a longer exponential
128
+ // backoff between the 4 retries (30s / 60s / 120s); api surfaces keep the tighter
129
+ // defaults. An explicit caller timeoutMs still wins.
130
+ const policy = retryPolicyForSurface(surface);
131
+ const effTimeout = timeoutMs != null ? timeoutMs : policy.timeoutMs;
132
+ const baseDelayMs = policy.baseDelayMs;
133
+
134
+ const laneRunner = () => {
135
+ switch (surface) {
136
+ case 'api': return completeAnthropicApi({ system, prompt, model, maxTokens, temperature });
137
+ case 'claude-cli': return completeClaudeCli({ system, prompt, model, timeoutMs: effTimeout });
138
+ case 'openai-api': return completeOpenaiApi({ system, prompt, model, maxTokens, temperature, timeoutMs: effTimeout });
139
+ case 'openai-cli': return completeCodexCli({ system, prompt, model, timeoutMs: effTimeout });
140
+ default: throw new Error(`unknown surface: ${surface}`);
141
+ }
142
+ };
57
143
  // A published run makes ~1000s of calls; be patient with transient
58
- // throttling/cold-starts so one blip doesn't abort a multi-hour grind.
59
- const out = await withRetry(() => withTimeout(runner, timeoutMs, `provider(${surface})`), { tries: 4, baseDelayMs: 3000 });
60
- return { ...out, surface };
144
+ // throttling/cold-starts so one blip doesn't abort a multi-hour grind. `attempts`
145
+ // counts every try (retries included) so the budget can charge for them.
146
+ let attempts = 0;
147
+ const runner = () => { attempts += 1; return withTimeout(laneRunner, effTimeout, `provider(${surface})`); };
148
+ try {
149
+ const out = await withRetry(runner, { tries: policy.tries, baseDelayMs });
150
+ return { ...out, surface, provider, attempts };
151
+ } catch (e) {
152
+ // Surface the attempt count so a persistently-failing call can be charged for
153
+ // (and, for a timeout, marked failed_timeout by the runner instead of fatal).
154
+ if (e && typeof e === 'object') e.attempts = attempts;
155
+ throw e;
156
+ }
61
157
  }
62
158
 
63
- async function completeApi({ system, prompt, model, maxTokens, temperature }) {
159
+ // ── anthropic/api ─────────────────────────────────────────────────────────────
160
+ async function completeAnthropicApi({ system, prompt, model, maxTokens, temperature }) {
64
161
  let Anthropic;
65
162
  try { Anthropic = require('@anthropic-ai/sdk'); }
66
- catch (_e) { throw new Error('CLAUDE_PROVIDER=api requires the @anthropic-ai/sdk package (npm i @anthropic-ai/sdk)'); }
67
- if (!process.env.ANTHROPIC_API_KEY) throw new Error('CLAUDE_PROVIDER=api requires ANTHROPIC_API_KEY');
163
+ catch (_e) { throw new Error('the anthropic/api surface requires the @anthropic-ai/sdk package (npm i @anthropic-ai/sdk)'); }
164
+ if (!process.env.ANTHROPIC_API_KEY) throw new Error('the anthropic/api surface requires ANTHROPIC_API_KEY');
68
165
  const client = new Anthropic({ apiKey: process.env.ANTHROPIC_API_KEY });
69
166
  const params = {
70
167
  model: resolveModel(model),
@@ -84,7 +181,8 @@ async function completeApi({ system, prompt, model, maxTokens, temperature }) {
84
181
  };
85
182
  }
86
183
 
87
- function completeCli({ system, prompt, model, timeoutMs }) {
184
+ // ── anthropic/cli (claude -p) ──────────────────────────────────────────────────
185
+ function completeClaudeCli({ system, prompt, model, timeoutMs }) {
88
186
  return new Promise((resolve, reject) => {
89
187
  // Strip ANTHROPIC_API_KEY so the CLI uses the subscription session, not the
90
188
  // metered API key. Everything else in the env is preserved.
@@ -102,7 +200,7 @@ function completeCli({ system, prompt, model, timeoutMs }) {
102
200
  child.stderr.on('data', (d) => { err += d; });
103
201
  child.on('error', (e) => {
104
202
  clearTimeout(killer);
105
- if (e.code === 'ENOENT') reject(new Error("CLAUDE_PROVIDER=cli requires the `claude` CLI on PATH. Install it or set CLAUDE_PROVIDER=api."));
203
+ if (e.code === 'ENOENT') reject(new Error("the anthropic/cli surface requires the `claude` CLI on PATH. Install it or set CLAUDE_PROVIDER=api."));
106
204
  else reject(e);
107
205
  });
108
206
  child.on('close', (code) => {
@@ -115,4 +213,149 @@ function completeCli({ system, prompt, model, timeoutMs }) {
115
213
  });
116
214
  }
117
215
 
118
- module.exports = { complete, resolveModel, surfaceLabel, providerName, MODEL_ALIASES };
216
+ // ── openai/api (Chat Completions-compatible) ───────────────────────────────────
217
+ // Generic OpenAI-compatible client using global fetch (Node ≥ 18). The base_url
218
+ // is configurable — env OPENAI_BASE_URL wins, else the registry's provider
219
+ // base_url (lib/models.providerConfig), else api.openai.com/v1. Building it
220
+ // base_url-first is deliberate: a future provider is a registry entry with a
221
+ // base_url + api-key env, not a new code path.
222
+ async function completeOpenaiApi({ system, prompt, model, maxTokens, temperature, timeoutMs }) {
223
+ const cfg = openaiProviderConfig();
224
+ const apiKey = process.env[cfg.apiKeyEnv] || process.env.OPENAI_API_KEY;
225
+ if (!apiKey) throw new Error(`the openai/api surface requires ${cfg.apiKeyEnv} (or run the codex CLI surface). Set OPENAI_SURFACE=cli to use the subscription instead.`);
226
+ const baseUrl = (process.env.OPENAI_BASE_URL || cfg.baseUrl || 'https://api.openai.com/v1').replace(/\/+$/, '');
227
+ const messages = [];
228
+ if (system) messages.push({ role: 'system', content: system });
229
+ messages.push({ role: 'user', content: prompt });
230
+ const body = { model: resolveModel(model), messages };
231
+ // Newer OpenAI models use `max_completion_tokens`; classic Chat Completions use
232
+ // `max_tokens`. Send the modern field; harmless on classic-compatible servers
233
+ // that ignore unknown fields, and correct for current OpenAI models.
234
+ if (maxTokens) body.max_completion_tokens = maxTokens;
235
+ if (temperature !== undefined) body.temperature = temperature;
236
+
237
+ const controller = new AbortController();
238
+ const t = setTimeout(() => controller.abort(), timeoutMs);
239
+ let resp;
240
+ try {
241
+ resp = await fetch(`${baseUrl}/chat/completions`, {
242
+ method: 'POST',
243
+ headers: { 'content-type': 'application/json', authorization: `Bearer ${apiKey}` },
244
+ body: JSON.stringify(body),
245
+ signal: controller.signal,
246
+ });
247
+ } finally { clearTimeout(t); }
248
+ const raw = await resp.text();
249
+ if (!resp.ok) throw new Error(`openai/api ${resp.status}: ${raw.slice(0, 300)}`);
250
+ let parsed;
251
+ try { parsed = JSON.parse(raw); } catch (_e) { throw new Error(`openai/api: unparseable response: ${raw.slice(0, 200)}`); }
252
+ const text = ((parsed.choices && parsed.choices[0] && parsed.choices[0].message && parsed.choices[0].message.content) || '');
253
+ const usage = parsed.usage || {};
254
+ return {
255
+ text: String(text).trim(),
256
+ usage: {
257
+ input_tokens: usage.prompt_tokens || 0,
258
+ output_tokens: usage.completion_tokens || 0,
259
+ },
260
+ };
261
+ }
262
+
263
+ // Read the OpenAI provider config (base_url + api-key env) from the registry,
264
+ // lazily (require inside the fn) so provider.js keeps no load-time cycle with
265
+ // lib/models.js. Falls back to the public defaults if the registry omits it.
266
+ function openaiProviderConfig() {
267
+ try {
268
+ const cfg = require('./models').providerConfig('openai');
269
+ if (cfg) return { baseUrl: cfg.base_url, apiKeyEnv: cfg.api_key_env || 'OPENAI_API_KEY' };
270
+ } catch (_e) { /* fall through to defaults */ }
271
+ return { baseUrl: 'https://api.openai.com/v1', apiKeyEnv: 'OPENAI_API_KEY' };
272
+ }
273
+
274
+ // ── openai/cli (codex exec — ChatGPT subscription) ─────────────────────────────
275
+ // Invocation template (from the verified recon, reference_codex_cli):
276
+ // codex exec --json -s read-only --skip-git-repo-check --ephemeral -o <tmp> [-m <model>] "PROMPT"
277
+ // Honoured facts:
278
+ // - `-a/--ask-for-approval` is INVALID on `codex exec` — it is NEVER passed;
279
+ // exec already defaults to approval_policy "never".
280
+ // - the model id is NOT echoed in the JSONL stream; we set -m so the receipt
281
+ // records what WE requested (run.model_id).
282
+ // - auth lives at ~/.codex/auth.json; we do a PRESENCE check only and NEVER
283
+ // read or print its contents.
284
+ // - `codex exec` has no system-prompt flag, so a system prompt (the SKILL.md in
285
+ // with_skill mode) is folded into the prompt text.
286
+ // - stderr emits a harmless "Reading additional input from stdin…" notice.
287
+ const CODEX_EXEC_ARGS = ['exec', '--json', '-s', 'read-only', '--skip-git-repo-check', '--ephemeral'];
288
+
289
+ function codexAuthPresent(homeDir) {
290
+ const home = homeDir || process.env.CODEX_HOME_DIR || os.homedir();
291
+ // CODEX_HOME overrides the auth location if the user set it (codex convention).
292
+ const base = process.env.CODEX_HOME || path.join(home, '.codex');
293
+ return fs.existsSync(path.join(base, 'auth.json'));
294
+ }
295
+
296
+ // Build the exact argv for a codex exec call. Exposed for the gate to assert the
297
+ // template (and that `-a` never appears). `outFile` receives the final message.
298
+ function buildCodexArgs({ model, outFile }) {
299
+ const args = [...CODEX_EXEC_ARGS, '-o', outFile];
300
+ if (model) args.push('-m', resolveModel(model));
301
+ return args;
302
+ }
303
+
304
+ // The FULL argv for a codex call: the template plus a single `-` positional that
305
+ // means "read the prompt from stdin". The prompt is NEVER a positional argv —
306
+ // real SKILL.md files start with `---` (YAML frontmatter) and codex rejects a
307
+ // positional beginning with `--`. Exposed so the gate can lock this in.
308
+ function codexFinalArgs({ model, outFile }) {
309
+ return [...buildCodexArgs({ model, outFile }), '-'];
310
+ }
311
+
312
+ function completeCodexCli({ system, prompt, model, timeoutMs }) {
313
+ return new Promise((resolve, reject) => {
314
+ if (!codexAuthPresent()) {
315
+ return reject(new Error('the openai/cli surface requires codex auth (~/.codex/auth.json). Run `codex login` (or `codex login --device-auth` on a headless box).'));
316
+ }
317
+ // codex exec has no system-prompt flag → fold the system prompt into the
318
+ // prompt text (this is how the SKILL.md reaches the model in with_skill mode).
319
+ const fullPrompt = system ? `${system}\n\n---\n\n${prompt}` : prompt;
320
+ const outFile = path.join(os.tmpdir(), `driftproof-codex-${process.pid}-${Date.now()}-${Math.floor(codexCounter())}.txt`);
321
+ // The prompt is delivered on STDIN, not as a positional argv: real SKILL.md
322
+ // files begin with `---` (YAML frontmatter), and a positional argument that
323
+ // starts with `--` is rejected by codex's arg parser ("unexpected argument
324
+ // '---'"). `-` as the positional tells `codex exec` to read instructions from
325
+ // stdin (per its --help), which is content-agnostic.
326
+ const args = codexFinalArgs({ model, outFile });
327
+
328
+ const child = spawn('codex', args, { env: { ...process.env }, stdio: ['pipe', 'ignore', 'pipe'] });
329
+ let err = '';
330
+ const killer = setTimeout(() => { child.kill('SIGKILL'); }, timeoutMs + 5000);
331
+ child.stderr.on('data', (d) => { err += d; });
332
+ child.on('error', (e) => {
333
+ clearTimeout(killer);
334
+ if (e.code === 'ENOENT') reject(new Error('the openai/cli surface requires the `codex` CLI on PATH (npm i -g @openai/codex).'));
335
+ else reject(e);
336
+ });
337
+ child.on('close', (code) => {
338
+ clearTimeout(killer);
339
+ let text = '';
340
+ try { text = fs.readFileSync(outFile, 'utf8'); } catch (_e) { /* fall through */ }
341
+ try { fs.unlinkSync(outFile); } catch (_e) { /* best effort */ }
342
+ if (code !== 0) return reject(new Error(`codex exec exited ${code}: ${String(err).slice(0, 400)}`));
343
+ resolve({ text: String(text).trim(), usage: null });
344
+ });
345
+ child.stdin.on('error', () => { /* ignore EPIPE if codex exits before reading all input */ });
346
+ child.stdin.write(fullPrompt);
347
+ child.stdin.end();
348
+ });
349
+ }
350
+
351
+ // A monotonically-increasing counter for temp-file uniqueness that does not use
352
+ // Math.random (kept deterministic-friendly for any harness that forbids it).
353
+ let _codexCounter = 0;
354
+ function codexCounter() { _codexCounter += 1; return _codexCounter; }
355
+
356
+ module.exports = {
357
+ complete, resolveModel, surfaceLabel, providerName, MODEL_ALIASES,
358
+ inferProvider, surfaceForModel, providerForSurface,
359
+ isMeteredSurface, isSubscriptionSurface, retryPolicyForSurface,
360
+ CODEX_OVERHEAD_NOTE, CODEX_EXEC_ARGS, buildCodexArgs, codexFinalArgs, codexAuthPresent,
361
+ };
package/lib/receipt.js CHANGED
@@ -13,7 +13,8 @@ const { aggregateBands, combineUncertainty, round } = require('./stats');
13
13
  const SCHEMA_FILES = {
14
14
  '0.1': 'receipt.v0.1.schema.json',
15
15
  '0.2': 'receipt.v0.2.schema.json',
16
- '0.3': 'receipt.schema.json',
16
+ '0.3': 'receipt.v0.3.schema.json',
17
+ '0.3.1': 'receipt.schema.json',
17
18
  };
18
19
 
19
20
  const _validators = {};
@@ -80,8 +81,12 @@ function aggregate(caseResults) {
80
81
  // cases: [ { id, mode, outcome, score, mean, stddev, samples, generation_hash, judge_sample_hashes, threshold, reason, judge } ]
81
82
  // editorialReviews: optional [ { url, source, date } ]
82
83
  function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED', editorialReviews = null }) {
83
- const withSkill = cases.filter((c) => c.mode === 'with_skill');
84
- const baseline = cases.filter((c) => c.mode === 'baseline');
84
+ // v0.3.1: cases marked failed_timeout are recorded in results.cases but EXCLUDED
85
+ // from aggregates — a band is never fabricated from a case that did not complete.
86
+ const okCases = cases.filter((c) => c.case_status !== 'failed_timeout');
87
+ const failedCount = cases.length - okCases.length;
88
+ const withSkill = okCases.filter((c) => c.mode === 'with_skill');
89
+ const baseline = okCases.filter((c) => c.mode === 'baseline');
85
90
  const aggWith = aggregate(withSkill);
86
91
  const aggBase = aggregate(baseline);
87
92
 
@@ -100,6 +105,9 @@ function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED',
100
105
  run: {
101
106
  model_id: run.model_id,
102
107
  model_release_date: run.model_release_date == null ? null : run.model_release_date,
108
+ // v0.3.1: two-axis provider (registry `provider`, else inferred). Defaults
109
+ // to anthropic for any legacy caller that omits it.
110
+ provider: run.provider || 'anthropic',
103
111
  surface: run.surface,
104
112
  runner_version: run.runner_version,
105
113
  date_utc: run.date_utc,
@@ -124,6 +132,17 @@ function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED',
124
132
  verification_level: verificationLevel,
125
133
  receipt_hash: '',
126
134
  };
135
+ // v0.3.1 additive-optional fields (canonicalization sorts keys, so placement
136
+ // here does not affect the hash):
137
+ // run.surface_overhead_note — the fixed harness preamble on the openai/cli surface.
138
+ // skill.tokens — estimated SKILL.md token size (value-per-token axis).
139
+ if (run.surface_overhead_note) receipt.run.surface_overhead_note = run.surface_overhead_note;
140
+ if (skill.tokens != null) receipt.skill.tokens = skill.tokens;
141
+ // v0.3.1: mark the receipt incomplete when any case failed (excluded above).
142
+ if (failedCount > 0) {
143
+ receipt.run.status = 'incomplete';
144
+ receipt.run.failed_case_count = failedCount;
145
+ }
127
146
  if (editorialReviews && editorialReviews.length) receipt.editorial_reviews = editorialReviews;
128
147
  return sealReceipt(receipt);
129
148
  }
package/lib/run.js CHANGED
@@ -1,12 +1,14 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
- const { complete, resolveModel, surfaceLabel } = require('./provider');
4
+ const { complete, resolveModel, surfaceForModel, CODEX_OVERHEAD_NOTE } = require('./provider');
5
5
  const { gradeSamples, judgeSettings } = require('./judge');
6
6
  const { buildReceipt } = require('./receipt');
7
7
  const { sha256 } = require('./canonical');
8
- const { registryStatus } = require('./models');
8
+ const { registryStatus, providerForModel } = require('./models');
9
9
  const { perCallCostUSD } = require('./cost');
10
+ const { runChecks } = require('./checks');
11
+ const { estimateTokens } = require('./skillCost');
10
12
  const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES } = require('../config');
11
13
 
12
14
  // Known model release dates (best-effort; null when unknown). Recorded into the
@@ -20,6 +22,8 @@ const MODEL_RELEASE_DATES = {
20
22
  'claude-haiku-4-5': '2025-10-01',
21
23
  'claude-sonnet-5': '2026-06-30', // anthropic.com/news/claude-sonnet-5
22
24
  'claude-sonnet-4-6': '2026-02-17', // anthropic.com/news/claude-sonnet-4-6
25
+ 'claude-opus-5': '2026-07-24', // anthropic.com/news/claude-opus-5
26
+ 'claude-opus-4-8': '2026-05-28', // anthropic.com/news/claude-opus-4-8
23
27
  };
24
28
 
25
29
  function releaseDateFor(modelId) {
@@ -39,9 +43,16 @@ function projectCalls(caseCount, samples) {
39
43
  // SKILL.md is prepended as a system prompt (the whole point: measure the skill's
40
44
  // marginal effect vs a bare baseline).
41
45
  async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs }) {
46
+ // Test seam (gate only): force a persistent timeout for a named case id so the
47
+ // failed_timeout path is exercised deterministically without any live call.
48
+ if (process.env.DRIFTPROOF_TEST_TIMEOUT_CASEID && process.env.DRIFTPROOF_TEST_TIMEOUT_CASEID === caseObj.id) {
49
+ const e = new Error('provider timed out (test seam) after retries');
50
+ e.code = 'TIMEOUT'; e.attempts = 4;
51
+ throw e;
52
+ }
42
53
  const system = withSkill ? skillMd : undefined;
43
- const { text, usage } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
44
- return { text, usage };
54
+ const { text, usage, attempts } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
55
+ return { text, usage, attempts };
45
56
  }
46
57
 
47
58
  // Determine a case outcome from its sampled band and threshold.
@@ -77,7 +88,10 @@ async function judgeCase({ caseObj, response, generationHash, judgeModel, mode,
77
88
  reason: g.reason,
78
89
  judge: { model_id: g.model_id, rubric_hash: g.rubric_hash },
79
90
  };
80
- return { caseResult, sampleTexts: g.sample_texts };
91
+ // v0.3.1 deterministic post-checks (supplementary; NOT folded into `outcome`).
92
+ const checks = runChecks(response, caseObj.checks);
93
+ if (checks.length) caseResult.checks = checks;
94
+ return { caseResult, sampleTexts: g.sample_texts, attempts: g.attempts };
81
95
  }
82
96
 
83
97
  // Run up to `concurrency` async tasks at a time, preserving input order in the
@@ -112,6 +126,9 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
112
126
  const modelId = resolveModel(model);
113
127
  const judgeModel = opts.judgeModel ? resolveModel(opts.judgeModel) : modelId;
114
128
  const timeoutMs = opts.timeoutMs || 120000;
129
+ // Optional per-case timeout overrides { caseId: ms }; a slow case can get a
130
+ // longer budget without lengthening every other case's per-call timeout.
131
+ const caseTimeoutMs = opts.caseTimeoutMs || {};
115
132
  const maxCalls = opts.maxCalls || 200;
116
133
  const samples = opts.samples || DEFAULT_JUDGE_SAMPLES;
117
134
  const concurrency = Math.max(1, opts.concurrency || 1);
@@ -137,46 +154,75 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
137
154
  for (const c of cases) for (const withSkill of [true, false]) tasks.push({ c, withSkill });
138
155
 
139
156
  let calls = 0;
157
+ let failedCases = 0;
158
+ const isTimeout = (e) => !!(e && (e.code === 'TIMEOUT' || /tim(e|ed)\s*out/i.test(String((e && e.message) || ''))));
159
+ const genKind = (ws) => (ws ? 'gen_with_skill' : 'gen_baseline');
140
160
  const pairs = await mapPool(tasks, concurrency, async ({ c, withSkill }) => {
141
161
  const mode = withSkill ? 'with_skill' : 'baseline';
142
- onProgress({ case: c.id, mode, phase: 'generate' });
143
- const { text } = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs });
144
- calls += 1;
145
- // Live budget: count the generation call, then hard-stop if over 1.25× cap.
146
- if (budget) budget.add(perCallCostUSD(modelId, withSkill ? 'gen_with_skill' : 'gen_baseline'));
147
- const generationHash = sha256(String(text || ''));
148
- onProgress({ case: c.id, mode, phase: 'judge', samples });
149
- const { caseResult, sampleTexts } = await judgeCase({ caseObj: c, response: text, generationHash, judgeModel, mode, timeoutMs, samples });
150
- calls += samples;
151
- // Live budget: count all `samples` judge calls for this (case, mode).
152
- if (budget) budget.add(samples * perCallCostUSD(judgeModel, 'judge'));
153
- onProgress({ case: c.id, mode, phase: 'done', outcome: caseResult.outcome, score: caseResult.mean, stddev: caseResult.stddev });
154
- const transcript = keepTranscripts
155
- ? { id: c.id, mode, generation: String(text || ''), judge_outputs: sampleTexts }
156
- : null;
157
- return { caseResult, transcript };
162
+ const ct = caseTimeoutMs[c.id] || timeoutMs;
163
+ try {
164
+ onProgress({ case: c.id, mode, phase: 'generate' });
165
+ const gen = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs: ct });
166
+ calls += 1;
167
+ // Live budget: count the generation call INCLUDING retries, then hard-stop
168
+ // if over 1.25× cap.
169
+ if (budget) budget.add((gen.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
170
+ const generationHash = sha256(String(gen.text || ''));
171
+ onProgress({ case: c.id, mode, phase: 'judge', samples });
172
+ const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ct, samples });
173
+ calls += samples;
174
+ // Live budget: count all judge calls (retries included) for this (case, mode).
175
+ if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judgeModel, 'judge'));
176
+ onProgress({ case: c.id, mode, phase: 'done', outcome: jr.caseResult.outcome, score: jr.caseResult.mean, stddev: jr.caseResult.stddev });
177
+ const transcript = keepTranscripts
178
+ ? { id: c.id, mode, generation: String(gen.text || ''), judge_outputs: jr.sampleTexts }
179
+ : null;
180
+ return { caseResult: jr.caseResult, transcript };
181
+ } catch (e) {
182
+ if (e && e.code === 'BUDGET_HARDSTOP') throw e; // budget hard-stop stays fatal
183
+ if (!isTimeout(e)) throw e; // non-timeout errors stay fatal
184
+ // Persistent timeout → NON-FATAL: charge the consumed attempts, record the
185
+ // case as failed_timeout (no fabricated samples), and continue the run.
186
+ if (budget) {
187
+ try {
188
+ if (e.phase === 'judge') budget.add((e.judgeAttempts || 1) * perCallCostUSD(judgeModel, 'judge'));
189
+ else budget.add((e.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
190
+ } catch (be) { if (be && be.code === 'BUDGET_HARDSTOP') throw be; }
191
+ }
192
+ failedCases += 1;
193
+ onProgress({ case: c.id, mode, phase: 'failed', reason: String((e && e.message) || 'timeout') });
194
+ return { caseResult: { id: c.id, mode, case_status: 'failed_timeout', reason: String((e && e.message) || 'timeout').slice(0, 200) }, transcript: null };
195
+ }
158
196
  });
159
197
  const caseResults = pairs.map((p) => p.caseResult);
160
198
  const transcripts = keepTranscripts ? pairs.map((p) => p.transcript) : null;
161
199
 
200
+ const surface = surfaceForModel(modelId);
162
201
  const receipt = buildReceipt({
163
- skill: { name: skill.name, version: skill.version, contentHash: skill.contentHash },
202
+ skill: {
203
+ name: skill.name, version: skill.version, contentHash: skill.contentHash,
204
+ // v0.3.1 value-per-token axis: estimated SKILL.md token size.
205
+ tokens: estimateTokens(skill.skillMd),
206
+ },
164
207
  suite: { format: skill.suite.format, suiteHash: skill.suite.suiteHash, caseCount: skill.suite.caseCount },
165
208
  run: {
166
209
  model_id: modelId,
167
210
  model_release_date: releaseDateFor(modelId),
168
- surface: surfaceLabel(),
211
+ provider: providerForModel(modelId),
212
+ surface,
213
+ // v0.3.1: on the openai/cli (codex) surface, record the fixed harness preamble.
214
+ surface_overhead_note: surface === 'openai-cli' ? CODEX_OVERHEAD_NOTE : undefined,
169
215
  runner_version: RUNNER_VERSION,
170
216
  date_utc: opts.nowIso || new Date().toISOString(),
171
217
  registry: registryStatus(modelId),
172
218
  transcripts: keepTranscripts ? 'retained-local' : 'hashes-only',
173
- judge: judgeSettings(samples),
219
+ judge: judgeSettings(samples, judgeModel),
174
220
  },
175
221
  cases: caseResults,
176
222
  verificationLevel: 'TESTED',
177
223
  });
178
224
 
179
- return { receipt, calls, transcripts };
225
+ return { receipt, calls, transcripts, failedCases };
180
226
  }
181
227
 
182
228
  function band(mean, sd) { return `${mean.toFixed(3)} ± ${sd.toFixed(3)}`; }
@@ -191,7 +237,7 @@ function summarizeReceipt(receipt) {
191
237
  L.push(`- **run (UTC):** ${receipt.run.date_utc}`);
192
238
  L.push(`- **runner:** v${receipt.run.runner_version}`);
193
239
  const j = receipt.run.judge || {};
194
- L.push(`- **judge:** ${j.samples || 1} samples/case, temperature ${j.temperature == null ? 'n/a (surface-controlled)' : j.temperature} (${j.sampling || 'single'})`);
240
+ L.push(`- **judge:** ${j.samples || 1} samples/case, temperature ${j.temperature == null ? 'n/a' : j.temperature} (${j.sampling || 'single'})`);
195
241
  if (receipt.run.registry) L.push(`- **registry:** ${receipt.run.registry} **transcripts:** ${receipt.run.transcripts || 'hashes-only'}`);
196
242
  L.push(`- **skill content_hash:** \`${receipt.skill.content_hash.slice(0, 16)}…\``);
197
243
  L.push(`- **suite:** ${receipt.suite.case_count} cases (${receipt.suite.format})`);
@@ -203,6 +249,10 @@ function summarizeReceipt(receipt) {
203
249
  const cmp = receipt.comparison;
204
250
  const aggs = receipt.results.aggregates;
205
251
  const sign = cmp.delta >= 0 ? '+' : '';
252
+ if (receipt.run.status === 'incomplete') {
253
+ L.push(`> ⏱ **INCOMPLETE** — ${receipt.run.failed_case_count} case(s) failed (timed out after retries) and are EXCLUDED from the aggregates below; this receipt must not be used to compute a drift/durability verdict.`);
254
+ L.push('');
255
+ }
206
256
  L.push(`with_skill **${band(aggs.with_skill.mean_score, aggs.with_skill.stddev)}** vs baseline **${band(aggs.baseline.mean_score, aggs.baseline.stddev)}**`);
207
257
  L.push('');
208
258
  L.push(`skill lift **${sign}${cmp.delta.toFixed(3)}** (combined uncertainty ± ${cmp.delta_uncertainty.toFixed(3)})`);
@@ -212,6 +262,10 @@ function summarizeReceipt(receipt) {
212
262
  L.push(`| case | mode | outcome | mean ± stddev | judge reason |`);
213
263
  L.push(`|---|---|---|---|---|`);
214
264
  for (const c of receipt.results.cases) {
265
+ if (c.case_status === 'failed_timeout') {
266
+ L.push(`| \`${c.id}\` | ${c.mode} | ⏱ failed_timeout | — (not measured) | ${c.reason || 'timed out'} |`);
267
+ continue;
268
+ }
215
269
  const flag = c.outcome === 'borderline' ? ' ⚠' : '';
216
270
  L.push(`| \`${c.id}\` | ${c.mode} | ${c.outcome}${flag} | ${band(c.mean, c.stddev || 0)} | ${c.reason || ''} |`);
217
271
  }