driftproof 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/provider.js CHANGED
@@ -1,20 +1,36 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
+ const fs = require('fs');
5
+ const os = require('os');
6
+ const path = require('path');
4
7
  const { spawn } = require('child_process');
5
8
  const { withRetry, withTimeout } = require('./json');
6
9
  const { stubComplete, stubEnabled } = require('./stub');
10
+ const {
11
+ parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage, normalizeUsage,
12
+ } = require('./usage');
7
13
 
8
- // Provider abstraction: one `complete()` call, two surfaces.
14
+ // Provider abstraction: one `complete()` call over a TWO-AXIS lane model —
15
+ // provider (anthropic | openai) × surface (api | cli). Four concrete lanes:
9
16
  //
10
- // CLAUDE_PROVIDER=api → Anthropic Messages API (needs ANTHROPIC_API_KEY).
11
- // CLAUDE_PROVIDER=cli → spawn `claude -p` with ANTHROPIC_API_KEY STRIPPED
12
- // from the child env, so dev runs draw on the local
13
- // subscription session credit rather than metered API
14
- // billing. This is the DEFAULT for dev.
17
+ // anthropic/api → Anthropic Messages API (ANTHROPIC_API_KEY) label "api"
18
+ // anthropic/cli → spawn `claude -p` on the subscription label "claude-cli"
19
+ // openai/api → Chat Completions-compatible API (OPENAI_API_KEY, base_url
20
+ // configurable — generic so a future provider is just a
21
+ // registry entry with a base_url) label "openai-api"
22
+ // openai/cli → spawn `codex exec` on the ChatGPT subscription label "openai-cli"
23
+ //
24
+ // The provider of a call is inferred from the MODEL id (so a run can generate on
25
+ // an OpenAI model while the fixed Haiku judge still runs on Anthropic). The
26
+ // surface for each provider is chosen from the environment:
27
+ //
28
+ // anthropic: CLAUDE_PROVIDER = api | cli (default cli — subscription)
29
+ // openai: OPENAI_SURFACE = api | cli (default: api iff OPENAI_API_KEY
30
+ // is present, else cli/codex)
15
31
  //
16
32
  // The surface actually used is returned so the runner can stamp it into the
17
- // receipt (run.surface = "api" | "claude-cli").
33
+ // receipt (run.surface), together with run.provider.
18
34
 
19
35
  // Short model aliases → canonical ids. Kept tiny and explicit; unknown values
20
36
  // are passed through verbatim so a full model id always works.
@@ -24,47 +40,153 @@ const MODEL_ALIASES = {
24
40
  'sonnet-5': 'claude-sonnet-5',
25
41
  'sonnet-4-6': 'claude-sonnet-4-6', // previous Sonnet point release
26
42
  opus: 'claude-opus-4-8',
43
+ // OpenAI convenience aliases (full ids always work too).
44
+ gpt: 'gpt-5.6-sol',
45
+ 'gpt-flagship': 'gpt-5.6-sol',
27
46
  };
28
47
 
29
48
  function resolveModel(m) {
30
49
  return MODEL_ALIASES[m] || m;
31
50
  }
32
51
 
52
+ // Infer the provider from a (resolved) model id. Pure and self-contained (no
53
+ // registry require) so provider.js has no cycle with lib/models.js; the registry
54
+ // `provider` field is authoritative where a model is registered (see
55
+ // lib/models.providerForModel), and registered OpenAI ids match this prefix too.
56
+ function inferProvider(modelId) {
57
+ const id = String(resolveModel(modelId) || '').toLowerCase();
58
+ if (/^(gpt-|o[1-9]|chatgpt|codex|text-|davinci|omni)/.test(id)) return 'openai';
59
+ return 'anthropic';
60
+ }
61
+
62
+ // ── surface selection per provider ───────────────────────────────────────────
63
+ function anthropicSurface() {
64
+ return (process.env.CLAUDE_PROVIDER || 'cli').toLowerCase() === 'api' ? 'api' : 'claude-cli';
65
+ }
66
+
67
+ function openaiSurface() {
68
+ const pref = (process.env.OPENAI_SURFACE || '').toLowerCase();
69
+ if (pref === 'api') return 'openai-api';
70
+ if (pref === 'cli' || pref === 'codex') return 'openai-cli';
71
+ // Auto: prefer the metered API when a key is present (published-run default),
72
+ // else fall back to the Codex subscription surface.
73
+ return process.env.OPENAI_API_KEY ? 'openai-api' : 'openai-cli';
74
+ }
75
+
76
+ function surfaceForModel(modelId) {
77
+ return inferProvider(modelId) === 'openai' ? openaiSurface() : anthropicSurface();
78
+ }
79
+
80
+ function providerForSurface(surface) {
81
+ return (surface === 'openai-api' || surface === 'openai-cli') ? 'openai' : 'anthropic';
82
+ }
83
+
84
+ // Backward-compatible default surface label (the Anthropic axis). Kept so callers
85
+ // that stamp a surface without a model (legacy) keep their prior meaning.
86
+ function surfaceLabel() { return anthropicSurface(); }
87
+
33
88
  function providerName() {
34
89
  return (process.env.CLAUDE_PROVIDER || 'cli').toLowerCase();
35
90
  }
36
91
 
37
- // The receipt-facing label for the surface in use.
38
- function surfaceLabel() {
39
- return providerName() === 'api' ? 'api' : 'claude-cli';
40
- }
41
-
42
- // Send a single-turn prompt and return { text, usage, surface }.
43
- // usage is { input_tokens, output_tokens } when the surface reports it (api),
44
- // else null (cli does not expose token counts to us).
45
- // `temperature` is honoured ONLY on the api surface (the Messages API exposes
46
- // it). On the cli surface sampling params are surface-controlled — we cannot set
47
- // them, so temperature is ignored there and the receipt records that fact.
48
- async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = 120000, temperature = undefined }) {
49
- const surface = surfaceLabel();
50
- // Offline stub surface: return a canned completion with zero model calls. The
51
- // receipt still records the real surface label so a stub run is not mistaken
52
- // for a genuine one at read time — only run.judge/generation TEXT is canned.
53
- if (stubEnabled()) return { ...stubComplete({ system, prompt }), surface };
54
- const runner = () => (surface === 'api'
55
- ? completeApi({ system, prompt, model, maxTokens, temperature })
56
- : completeCli({ system, prompt, model, timeoutMs }));
92
+ // A surface whose spend is metered (real dollars) vs a subscription surface
93
+ // (metered spend $0; the $ figure is the estimated-equivalent API cost).
94
+ function isMeteredSurface(surface) { return surface === 'api' || surface === 'openai-api'; }
95
+ function isSubscriptionSurface(surface) { return !isMeteredSurface(surface); }
96
+
97
+ // Per-surface retry/timeout policy. A cli/subscription surface spawns a first-party
98
+ // CLI subprocess (cold-start-dominated, occasionally hangs), so it gets a longer
99
+ // per-call timeout (300s) and a longer exponential backoff (baseDelay 30s → 30/60/
100
+ // 120s across the 4 retries); api surfaces keep the tighter 120s / 3s defaults.
101
+ function retryPolicyForSurface(surface) {
102
+ const cli = isSubscriptionSurface(surface);
103
+ return { timeoutMs: cli ? 300000 : 120000, baseDelayMs: cli ? 30000 : 3000, tries: 4 };
104
+ }
105
+
106
+ // The fixed harness preamble Codex prepends to every `codex exec` call. Recorded
107
+ // into openai/cli receipts as run.surface_overhead_note so a reader knows the
108
+ // per-call input-token count is dominated by a constant we do not control.
109
+ const CODEX_OVERHEAD_NOTE =
110
+ 'openai/cli surface (codex exec): every call carries a fixed Codex base-instruction '
111
+ + 'preamble of ~12,000–15,000 input tokens that the harness prepends and we do not '
112
+ + 'control; the model id is set by us via -m (it is not echoed in the JSONL stream). '
113
+ + 'Approval prompting is off by default on `codex exec` (no -a flag is passed).';
114
+
115
+ // Send a single-turn prompt and return { text, usage, wall_ms, surface, provider }.
116
+ //
117
+ // usage is the normalized v0.4 record { input_tokens, output_tokens, cached_tokens,
118
+ // wall_ms } on EVERY lane — including the two CLI lanes, which report it in their
119
+ // structured output (`claude -p --output-format json`; the `codex exec --json`
120
+ // JSONL stream). See lib/usage.js for the per-surface shapes and the input-token
121
+ // normalization. Fields the surface does not report stay null, never 0.
122
+ //
123
+ // wall_ms is measured HERE, around the successful attempt, so it means the same
124
+ // thing on all four lanes (retries are excluded — a retried call's latency would
125
+ // describe our backoff, not the model).
126
+ //
127
+ // `temperature` is honoured ONLY on api surfaces; cli surfaces control sampling
128
+ // themselves, so temperature is ignored there and the receipt records that fact.
129
+ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = null, temperature = undefined }) {
130
+ const surface = surfaceForModel(model);
131
+ const provider = providerForSurface(surface);
132
+ // Offline stub surface: canned completion, zero model calls. The receipt still
133
+ // records the real surface/provider so a stub run is not mistaken for a genuine
134
+ // one at read time — only the generation/judge TEXT is canned.
135
+ if (stubEnabled()) {
136
+ const s = stubComplete({ system, prompt });
137
+ return { ...s, usage: normalizeUsage(s.usage), wall_ms: (s.usage && s.usage.wall_ms) || 0, surface, provider, attempts: 1 };
138
+ }
139
+
140
+ // Per-surface tolerance (see retryPolicyForSurface): a cli/subscription surface
141
+ // spawns a first-party CLI subprocess whose cold-start is slow and occasionally
142
+ // hangs, so it gets a longer per-call timeout (300s) and a longer exponential
143
+ // backoff between the 4 retries (30s / 60s / 120s); api surfaces keep the tighter
144
+ // defaults. An explicit caller timeoutMs still wins.
145
+ const policy = retryPolicyForSurface(surface);
146
+ const effTimeout = timeoutMs != null ? timeoutMs : policy.timeoutMs;
147
+ const baseDelayMs = policy.baseDelayMs;
148
+
149
+ const laneRunner = () => {
150
+ switch (surface) {
151
+ case 'api': return completeAnthropicApi({ system, prompt, model, maxTokens, temperature });
152
+ case 'claude-cli': return completeClaudeCli({ system, prompt, model, timeoutMs: effTimeout });
153
+ case 'openai-api': return completeOpenaiApi({ system, prompt, model, maxTokens, temperature, timeoutMs: effTimeout });
154
+ case 'openai-cli': return completeCodexCli({ system, prompt, model, timeoutMs: effTimeout });
155
+ default: throw new Error(`unknown surface: ${surface}`);
156
+ }
157
+ };
57
158
  // A published run makes ~1000s of calls; be patient with transient
58
- // throttling/cold-starts so one blip doesn't abort a multi-hour grind.
59
- const out = await withRetry(() => withTimeout(runner, timeoutMs, `provider(${surface})`), { tries: 4, baseDelayMs: 3000 });
60
- return { ...out, surface };
159
+ // throttling/cold-starts so one blip doesn't abort a multi-hour grind. `attempts`
160
+ // counts every try (retries included) so the budget can charge for them.
161
+ let attempts = 0;
162
+ // Wall-clock of the attempt that SUCCEEDED (each attempt overwrites, so a
163
+ // retried call reports the latency of the call that actually produced the text,
164
+ // not the accumulated backoff).
165
+ let wallMs = null;
166
+ const runner = () => {
167
+ attempts += 1;
168
+ const t0 = Date.now();
169
+ return withTimeout(laneRunner, effTimeout, `provider(${surface})`)
170
+ .then((r) => { wallMs = Date.now() - t0; return r; });
171
+ };
172
+ try {
173
+ const out = await withRetry(runner, { tries: policy.tries, baseDelayMs });
174
+ const usage = normalizeUsage({ ...(out.usage || {}), wall_ms: wallMs });
175
+ return { ...out, usage, wall_ms: wallMs, surface, provider, attempts };
176
+ } catch (e) {
177
+ // Surface the attempt count so a persistently-failing call can be charged for
178
+ // (and, for a timeout, marked failed_timeout by the runner instead of fatal).
179
+ if (e && typeof e === 'object') e.attempts = attempts;
180
+ throw e;
181
+ }
61
182
  }
62
183
 
63
- async function completeApi({ system, prompt, model, maxTokens, temperature }) {
184
+ // ── anthropic/api ─────────────────────────────────────────────────────────────
185
+ async function completeAnthropicApi({ system, prompt, model, maxTokens, temperature }) {
64
186
  let Anthropic;
65
187
  try { Anthropic = require('@anthropic-ai/sdk'); }
66
- catch (_e) { throw new Error('CLAUDE_PROVIDER=api requires the @anthropic-ai/sdk package (npm i @anthropic-ai/sdk)'); }
67
- if (!process.env.ANTHROPIC_API_KEY) throw new Error('CLAUDE_PROVIDER=api requires ANTHROPIC_API_KEY');
188
+ catch (_e) { throw new Error('the anthropic/api surface requires the @anthropic-ai/sdk package (npm i @anthropic-ai/sdk)'); }
189
+ if (!process.env.ANTHROPIC_API_KEY) throw new Error('the anthropic/api surface requires ANTHROPIC_API_KEY');
68
190
  const client = new Anthropic({ apiKey: process.env.ANTHROPIC_API_KEY });
69
191
  const params = {
70
192
  model: resolveModel(model),
@@ -75,23 +197,24 @@ async function completeApi({ system, prompt, model, maxTokens, temperature }) {
75
197
  if (temperature !== undefined) params.temperature = temperature;
76
198
  const resp = await client.messages.create(params);
77
199
  const text = (resp.content || []).filter((b) => b.type === 'text').map((b) => b.text).join('');
78
- return {
79
- text,
80
- usage: {
81
- input_tokens: (resp.usage && resp.usage.input_tokens) || 0,
82
- output_tokens: (resp.usage && resp.usage.output_tokens) || 0,
83
- },
84
- };
200
+ return { text, usage: parseAnthropicApiUsage(resp.usage) };
85
201
  }
86
202
 
87
- function completeCli({ system, prompt, model, timeoutMs }) {
203
+ // ── anthropic/cli (claude -p) ──────────────────────────────────────────────────
204
+ function completeClaudeCli({ system, prompt, model, timeoutMs }) {
88
205
  return new Promise((resolve, reject) => {
89
206
  // Strip ANTHROPIC_API_KEY so the CLI uses the subscription session, not the
90
207
  // metered API key. Everything else in the env is preserved.
91
208
  const env = { ...process.env };
92
209
  delete env.ANTHROPIC_API_KEY;
93
210
 
94
- const args = ['-p', '--model', resolveModel(model)];
211
+ // v0.4: `--output-format json` returns ONE JSON object carrying both the
212
+ // final text (`result`) and the token usage (`usage`) — the default text
213
+ // output carries no usage at all, which is why usage was previously null on
214
+ // this lane. The text is read from the parsed object; if the CLI ever emits
215
+ // something unparseable we fall back to the raw stdout so a run degrades to
216
+ // the old behaviour (text, no usage) rather than failing.
217
+ const args = ['-p', '--output-format', 'json', '--model', resolveModel(model)];
95
218
  if (system) args.push('--append-system-prompt', system);
96
219
 
97
220
  const child = spawn('claude', args, { env, stdio: ['pipe', 'pipe', 'pipe'] });
@@ -102,17 +225,167 @@ function completeCli({ system, prompt, model, timeoutMs }) {
102
225
  child.stderr.on('data', (d) => { err += d; });
103
226
  child.on('error', (e) => {
104
227
  clearTimeout(killer);
105
- if (e.code === 'ENOENT') reject(new Error("CLAUDE_PROVIDER=cli requires the `claude` CLI on PATH. Install it or set CLAUDE_PROVIDER=api."));
228
+ if (e.code === 'ENOENT') reject(new Error("the anthropic/cli surface requires the `claude` CLI on PATH. Install it or set CLAUDE_PROVIDER=api."));
106
229
  else reject(e);
107
230
  });
108
231
  child.on('close', (code) => {
109
232
  clearTimeout(killer);
110
233
  if (code !== 0) return reject(new Error(`claude CLI exited ${code}: ${err.slice(0, 400)}`));
111
- resolve({ text: out.trim(), usage: null });
234
+ const parsed = parseClaudeCliJson(out);
235
+ if (!parsed) return resolve({ text: out.trim(), usage: null });
236
+ if (parsed.isError) return reject(new Error(`claude CLI reported is_error: ${String(parsed.text || '').slice(0, 300)}`));
237
+ resolve({ text: String(parsed.text || '').trim(), usage: parsed.usage });
112
238
  });
113
239
  child.stdin.write(prompt);
114
240
  child.stdin.end();
115
241
  });
116
242
  }
117
243
 
118
- module.exports = { complete, resolveModel, surfaceLabel, providerName, MODEL_ALIASES };
244
+ // ── openai/api (Chat Completions-compatible) ───────────────────────────────────
245
+ // Generic OpenAI-compatible client using global fetch (Node ≥ 18). The base_url
246
+ // is configurable — env OPENAI_BASE_URL wins, else the registry's provider
247
+ // base_url (lib/models.providerConfig), else api.openai.com/v1. Building it
248
+ // base_url-first is deliberate: a future provider is a registry entry with a
249
+ // base_url + api-key env, not a new code path.
250
+ async function completeOpenaiApi({ system, prompt, model, maxTokens, temperature, timeoutMs }) {
251
+ const cfg = openaiProviderConfig();
252
+ const apiKey = process.env[cfg.apiKeyEnv] || process.env.OPENAI_API_KEY;
253
+ if (!apiKey) throw new Error(`the openai/api surface requires ${cfg.apiKeyEnv} (or run the codex CLI surface). Set OPENAI_SURFACE=cli to use the subscription instead.`);
254
+ const baseUrl = (process.env.OPENAI_BASE_URL || cfg.baseUrl || 'https://api.openai.com/v1').replace(/\/+$/, '');
255
+ const messages = [];
256
+ if (system) messages.push({ role: 'system', content: system });
257
+ messages.push({ role: 'user', content: prompt });
258
+ const body = { model: resolveModel(model), messages };
259
+ // Newer OpenAI models use `max_completion_tokens`; classic Chat Completions use
260
+ // `max_tokens`. Send the modern field; harmless on classic-compatible servers
261
+ // that ignore unknown fields, and correct for current OpenAI models.
262
+ if (maxTokens) body.max_completion_tokens = maxTokens;
263
+ if (temperature !== undefined) body.temperature = temperature;
264
+
265
+ const controller = new AbortController();
266
+ const t = setTimeout(() => controller.abort(), timeoutMs);
267
+ let resp;
268
+ try {
269
+ resp = await fetch(`${baseUrl}/chat/completions`, {
270
+ method: 'POST',
271
+ headers: { 'content-type': 'application/json', authorization: `Bearer ${apiKey}` },
272
+ body: JSON.stringify(body),
273
+ signal: controller.signal,
274
+ });
275
+ } finally { clearTimeout(t); }
276
+ const raw = await resp.text();
277
+ if (!resp.ok) throw new Error(`openai/api ${resp.status}: ${raw.slice(0, 300)}`);
278
+ let parsed;
279
+ try { parsed = JSON.parse(raw); } catch (_e) { throw new Error(`openai/api: unparseable response: ${raw.slice(0, 200)}`); }
280
+ const text = ((parsed.choices && parsed.choices[0] && parsed.choices[0].message && parsed.choices[0].message.content) || '');
281
+ return { text: String(text).trim(), usage: parseOpenaiApiUsage(parsed.usage) };
282
+ }
283
+
284
+ // Read the OpenAI provider config (base_url + api-key env) from the registry,
285
+ // lazily (require inside the fn) so provider.js keeps no load-time cycle with
286
+ // lib/models.js. Falls back to the public defaults if the registry omits it.
287
+ function openaiProviderConfig() {
288
+ try {
289
+ const cfg = require('./models').providerConfig('openai');
290
+ if (cfg) return { baseUrl: cfg.base_url, apiKeyEnv: cfg.api_key_env || 'OPENAI_API_KEY' };
291
+ } catch (_e) { /* fall through to defaults */ }
292
+ return { baseUrl: 'https://api.openai.com/v1', apiKeyEnv: 'OPENAI_API_KEY' };
293
+ }
294
+
295
+ // ── openai/cli (codex exec — ChatGPT subscription) ─────────────────────────────
296
+ // Invocation template (from the verified recon, reference_codex_cli):
297
+ // codex exec --json -s read-only --skip-git-repo-check --ephemeral -o <tmp> [-m <model>] "PROMPT"
298
+ // Honoured facts:
299
+ // - `-a/--ask-for-approval` is INVALID on `codex exec` — it is NEVER passed;
300
+ // exec already defaults to approval_policy "never".
301
+ // - the model id is NOT echoed in the JSONL stream; we set -m so the receipt
302
+ // records what WE requested (run.model_id).
303
+ // - auth lives at ~/.codex/auth.json; we do a PRESENCE check only and NEVER
304
+ // read or print its contents.
305
+ // - `codex exec` has no system-prompt flag, so a system prompt (the SKILL.md in
306
+ // with_skill mode) is folded into the prompt text.
307
+ // - stderr emits a harmless "Reading additional input from stdin…" notice.
308
+ const CODEX_EXEC_ARGS = ['exec', '--json', '-s', 'read-only', '--skip-git-repo-check', '--ephemeral'];
309
+
310
+ function codexAuthPresent(homeDir) {
311
+ const home = homeDir || process.env.CODEX_HOME_DIR || os.homedir();
312
+ // CODEX_HOME overrides the auth location if the user set it (codex convention).
313
+ const base = process.env.CODEX_HOME || path.join(home, '.codex');
314
+ return fs.existsSync(path.join(base, 'auth.json'));
315
+ }
316
+
317
+ // Build the exact argv for a codex exec call. Exposed for the gate to assert the
318
+ // template (and that `-a` never appears). `outFile` receives the final message.
319
+ function buildCodexArgs({ model, outFile }) {
320
+ const args = [...CODEX_EXEC_ARGS, '-o', outFile];
321
+ if (model) args.push('-m', resolveModel(model));
322
+ return args;
323
+ }
324
+
325
+ // The FULL argv for a codex call: the template plus a single `-` positional that
326
+ // means "read the prompt from stdin". The prompt is NEVER a positional argv —
327
+ // real SKILL.md files start with `---` (YAML frontmatter) and codex rejects a
328
+ // positional beginning with `--`. Exposed so the gate can lock this in.
329
+ function codexFinalArgs({ model, outFile }) {
330
+ return [...buildCodexArgs({ model, outFile }), '-'];
331
+ }
332
+
333
+ function completeCodexCli({ system, prompt, model, timeoutMs }) {
334
+ return new Promise((resolve, reject) => {
335
+ if (!codexAuthPresent()) {
336
+ return reject(new Error('the openai/cli surface requires codex auth (~/.codex/auth.json). Run `codex login` (or `codex login --device-auth` on a headless box).'));
337
+ }
338
+ // codex exec has no system-prompt flag → fold the system prompt into the
339
+ // prompt text (this is how the SKILL.md reaches the model in with_skill mode).
340
+ const fullPrompt = system ? `${system}\n\n---\n\n${prompt}` : prompt;
341
+ const outFile = path.join(os.tmpdir(), `driftproof-codex-${process.pid}-${Date.now()}-${Math.floor(codexCounter())}.txt`);
342
+ // The prompt is delivered on STDIN, not as a positional argv: real SKILL.md
343
+ // files begin with `---` (YAML frontmatter), and a positional argument that
344
+ // starts with `--` is rejected by codex's arg parser ("unexpected argument
345
+ // '---'"). `-` as the positional tells `codex exec` to read instructions from
346
+ // stdin (per its --help), which is content-agnostic.
347
+ const args = codexFinalArgs({ model, outFile });
348
+
349
+ // v0.4: stdout is CAPTURED (it was 'ignore'). `--json` streams JSONL events
350
+ // there, and the terminal `turn.completed` event carries this call's token
351
+ // usage — the only place codex reports it. The final message still comes from
352
+ // the -o file (cleaner than scraping the stream); the JSONL is read for usage.
353
+ const child = spawn('codex', args, { env: { ...process.env }, stdio: ['pipe', 'pipe', 'pipe'] });
354
+ let err = '';
355
+ let jsonl = '';
356
+ const killer = setTimeout(() => { child.kill('SIGKILL'); }, timeoutMs + 5000);
357
+ child.stdout.on('data', (d) => { jsonl += d; });
358
+ child.stderr.on('data', (d) => { err += d; });
359
+ child.on('error', (e) => {
360
+ clearTimeout(killer);
361
+ if (e.code === 'ENOENT') reject(new Error('the openai/cli surface requires the `codex` CLI on PATH (npm i -g @openai/codex).'));
362
+ else reject(e);
363
+ });
364
+ child.on('close', (code) => {
365
+ clearTimeout(killer);
366
+ let text = '';
367
+ try { text = fs.readFileSync(outFile, 'utf8'); } catch (_e) { /* fall through */ }
368
+ try { fs.unlinkSync(outFile); } catch (_e) { /* best effort */ }
369
+ if (code !== 0) return reject(new Error(`codex exec exited ${code}: ${String(err).slice(0, 400)}`));
370
+ const ev = parseCodexJsonl(jsonl);
371
+ // Prefer the -o file; fall back to the stream's agent_message if it is empty.
372
+ const finalText = String(text || ev.text || '').trim();
373
+ resolve({ text: finalText, usage: ev.usage });
374
+ });
375
+ child.stdin.on('error', () => { /* ignore EPIPE if codex exits before reading all input */ });
376
+ child.stdin.write(fullPrompt);
377
+ child.stdin.end();
378
+ });
379
+ }
380
+
381
+ // A monotonically-increasing counter for temp-file uniqueness that does not use
382
+ // Math.random (kept deterministic-friendly for any harness that forbids it).
383
+ let _codexCounter = 0;
384
+ function codexCounter() { _codexCounter += 1; return _codexCounter; }
385
+
386
+ module.exports = {
387
+ complete, resolveModel, surfaceLabel, providerName, MODEL_ALIASES,
388
+ inferProvider, surfaceForModel, providerForSurface,
389
+ isMeteredSurface, isSubscriptionSurface, retryPolicyForSurface,
390
+ CODEX_OVERHEAD_NOTE, CODEX_EXEC_ARGS, buildCodexArgs, codexFinalArgs, codexAuthPresent,
391
+ };
package/lib/receipt.js CHANGED
@@ -13,7 +13,9 @@ const { aggregateBands, combineUncertainty, round } = require('./stats');
13
13
  const SCHEMA_FILES = {
14
14
  '0.1': 'receipt.v0.1.schema.json',
15
15
  '0.2': 'receipt.v0.2.schema.json',
16
- '0.3': 'receipt.schema.json',
16
+ '0.3': 'receipt.v0.3.schema.json',
17
+ '0.3.1': 'receipt.v0.3.1.schema.json',
18
+ '0.4': 'receipt.schema.json',
17
19
  };
18
20
 
19
21
  const _validators = {};
@@ -79,9 +81,13 @@ function aggregate(caseResults) {
79
81
  // run: { model_id, model_release_date, surface, runner_version, date_utc, judge, registry, transcripts }
80
82
  // cases: [ { id, mode, outcome, score, mean, stddev, samples, generation_hash, judge_sample_hashes, threshold, reason, judge } ]
81
83
  // editorialReviews: optional [ { url, source, date } ]
82
- function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED', editorialReviews = null }) {
83
- const withSkill = cases.filter((c) => c.mode === 'with_skill');
84
- const baseline = cases.filter((c) => c.mode === 'baseline');
84
+ function buildReceipt({ skill, suite, run, cases, economics = null, verificationLevel = 'TESTED', editorialReviews = null }) {
85
+ // v0.3.1: cases marked failed_timeout are recorded in results.cases but EXCLUDED
86
+ // from aggregates — a band is never fabricated from a case that did not complete.
87
+ const okCases = cases.filter((c) => c.case_status !== 'failed_timeout');
88
+ const failedCount = cases.length - okCases.length;
89
+ const withSkill = okCases.filter((c) => c.mode === 'with_skill');
90
+ const baseline = okCases.filter((c) => c.mode === 'baseline');
85
91
  const aggWith = aggregate(withSkill);
86
92
  const aggBase = aggregate(baseline);
87
93
 
@@ -100,6 +106,9 @@ function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED',
100
106
  run: {
101
107
  model_id: run.model_id,
102
108
  model_release_date: run.model_release_date == null ? null : run.model_release_date,
109
+ // v0.3.1: two-axis provider (registry `provider`, else inferred). Defaults
110
+ // to anthropic for any legacy caller that omits it.
111
+ provider: run.provider || 'anthropic',
103
112
  surface: run.surface,
104
113
  runner_version: run.runner_version,
105
114
  date_utc: run.date_utc,
@@ -124,6 +133,22 @@ function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED',
124
133
  verification_level: verificationLevel,
125
134
  receipt_hash: '',
126
135
  };
136
+ // v0.3.1 additive-optional fields (canonicalization sorts keys, so placement
137
+ // here does not affect the hash):
138
+ // run.surface_overhead_note — the fixed harness preamble on the openai/cli surface.
139
+ // skill.tokens — estimated SKILL.md token size (value-per-token axis).
140
+ if (run.surface_overhead_note) receipt.run.surface_overhead_note = run.surface_overhead_note;
141
+ if (skill.tokens != null) receipt.skill.tokens = skill.tokens;
142
+ // v0.4 economics (additive-optional): the frozen prices this receipt's derived
143
+ // dollar figures were computed from, and the derived block itself. A receipt
144
+ // from a surface that reports no usage simply omits both.
145
+ if (run.pricing_snapshot) receipt.run.pricing_snapshot = run.pricing_snapshot;
146
+ if (economics) receipt.economics = economics;
147
+ // v0.3.1: mark the receipt incomplete when any case failed (excluded above).
148
+ if (failedCount > 0) {
149
+ receipt.run.status = 'incomplete';
150
+ receipt.run.failed_case_count = failedCount;
151
+ }
127
152
  if (editorialReviews && editorialReviews.length) receipt.editorial_reviews = editorialReviews;
128
153
  return sealReceipt(receipt);
129
154
  }