driftproof 0.3.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +63 -19
- package/bin/driftproof +66 -4
- package/config/models.json +14 -4
- package/config.js +55 -5
- package/lib/checks.js +50 -0
- package/lib/diff.js +28 -7
- package/lib/export.js +52 -0
- package/lib/importers.js +207 -0
- package/lib/judge.js +35 -13
- package/lib/models.js +58 -8
- package/lib/provider.js +318 -45
- package/lib/receipt.js +29 -4
- package/lib/run.js +108 -27
- package/lib/skill.js +12 -2
- package/lib/skillCost.js +31 -0
- package/lib/stub.js +24 -3
- package/lib/usage.js +168 -0
- package/lib/value.js +502 -0
- package/lib/verdict.js +10 -1
- package/package.json +1 -1
- package/spec/RECEIPT.md +116 -8
- package/spec/receipt.schema.json +809 -59
- package/spec/receipt.v0.3.1.schema.json +642 -0
- package/spec/receipt.v0.3.schema.json +211 -0
package/lib/provider.js
CHANGED
|
@@ -1,20 +1,36 @@
|
|
|
1
1
|
// SPDX-License-Identifier: Apache-2.0
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
+
const fs = require('fs');
|
|
5
|
+
const os = require('os');
|
|
6
|
+
const path = require('path');
|
|
4
7
|
const { spawn } = require('child_process');
|
|
5
8
|
const { withRetry, withTimeout } = require('./json');
|
|
6
9
|
const { stubComplete, stubEnabled } = require('./stub');
|
|
10
|
+
const {
|
|
11
|
+
parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage, normalizeUsage,
|
|
12
|
+
} = require('./usage');
|
|
7
13
|
|
|
8
|
-
// Provider abstraction: one `complete()` call
|
|
14
|
+
// Provider abstraction: one `complete()` call over a TWO-AXIS lane model —
|
|
15
|
+
// provider (anthropic | openai) × surface (api | cli). Four concrete lanes:
|
|
9
16
|
//
|
|
10
|
-
//
|
|
11
|
-
//
|
|
12
|
-
//
|
|
13
|
-
//
|
|
14
|
-
//
|
|
17
|
+
// anthropic/api → Anthropic Messages API (ANTHROPIC_API_KEY) label "api"
|
|
18
|
+
// anthropic/cli → spawn `claude -p` on the subscription label "claude-cli"
|
|
19
|
+
// openai/api → Chat Completions-compatible API (OPENAI_API_KEY, base_url
|
|
20
|
+
// configurable — generic so a future provider is just a
|
|
21
|
+
// registry entry with a base_url) label "openai-api"
|
|
22
|
+
// openai/cli → spawn `codex exec` on the ChatGPT subscription label "openai-cli"
|
|
23
|
+
//
|
|
24
|
+
// The provider of a call is inferred from the MODEL id (so a run can generate on
|
|
25
|
+
// an OpenAI model while the fixed Haiku judge still runs on Anthropic). The
|
|
26
|
+
// surface for each provider is chosen from the environment:
|
|
27
|
+
//
|
|
28
|
+
// anthropic: CLAUDE_PROVIDER = api | cli (default cli — subscription)
|
|
29
|
+
// openai: OPENAI_SURFACE = api | cli (default: api iff OPENAI_API_KEY
|
|
30
|
+
// is present, else cli/codex)
|
|
15
31
|
//
|
|
16
32
|
// The surface actually used is returned so the runner can stamp it into the
|
|
17
|
-
// receipt (run.surface
|
|
33
|
+
// receipt (run.surface), together with run.provider.
|
|
18
34
|
|
|
19
35
|
// Short model aliases → canonical ids. Kept tiny and explicit; unknown values
|
|
20
36
|
// are passed through verbatim so a full model id always works.
|
|
@@ -24,47 +40,153 @@ const MODEL_ALIASES = {
|
|
|
24
40
|
'sonnet-5': 'claude-sonnet-5',
|
|
25
41
|
'sonnet-4-6': 'claude-sonnet-4-6', // previous Sonnet point release
|
|
26
42
|
opus: 'claude-opus-4-8',
|
|
43
|
+
// OpenAI convenience aliases (full ids always work too).
|
|
44
|
+
gpt: 'gpt-5.6-sol',
|
|
45
|
+
'gpt-flagship': 'gpt-5.6-sol',
|
|
27
46
|
};
|
|
28
47
|
|
|
29
48
|
function resolveModel(m) {
|
|
30
49
|
return MODEL_ALIASES[m] || m;
|
|
31
50
|
}
|
|
32
51
|
|
|
52
|
+
// Infer the provider from a (resolved) model id. Pure and self-contained (no
|
|
53
|
+
// registry require) so provider.js has no cycle with lib/models.js; the registry
|
|
54
|
+
// `provider` field is authoritative where a model is registered (see
|
|
55
|
+
// lib/models.providerForModel), and registered OpenAI ids match this prefix too.
|
|
56
|
+
function inferProvider(modelId) {
|
|
57
|
+
const id = String(resolveModel(modelId) || '').toLowerCase();
|
|
58
|
+
if (/^(gpt-|o[1-9]|chatgpt|codex|text-|davinci|omni)/.test(id)) return 'openai';
|
|
59
|
+
return 'anthropic';
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
// ── surface selection per provider ───────────────────────────────────────────
|
|
63
|
+
function anthropicSurface() {
|
|
64
|
+
return (process.env.CLAUDE_PROVIDER || 'cli').toLowerCase() === 'api' ? 'api' : 'claude-cli';
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
function openaiSurface() {
|
|
68
|
+
const pref = (process.env.OPENAI_SURFACE || '').toLowerCase();
|
|
69
|
+
if (pref === 'api') return 'openai-api';
|
|
70
|
+
if (pref === 'cli' || pref === 'codex') return 'openai-cli';
|
|
71
|
+
// Auto: prefer the metered API when a key is present (published-run default),
|
|
72
|
+
// else fall back to the Codex subscription surface.
|
|
73
|
+
return process.env.OPENAI_API_KEY ? 'openai-api' : 'openai-cli';
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
function surfaceForModel(modelId) {
|
|
77
|
+
return inferProvider(modelId) === 'openai' ? openaiSurface() : anthropicSurface();
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
function providerForSurface(surface) {
|
|
81
|
+
return (surface === 'openai-api' || surface === 'openai-cli') ? 'openai' : 'anthropic';
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
// Backward-compatible default surface label (the Anthropic axis). Kept so callers
|
|
85
|
+
// that stamp a surface without a model (legacy) keep their prior meaning.
|
|
86
|
+
function surfaceLabel() { return anthropicSurface(); }
|
|
87
|
+
|
|
33
88
|
function providerName() {
|
|
34
89
|
return (process.env.CLAUDE_PROVIDER || 'cli').toLowerCase();
|
|
35
90
|
}
|
|
36
91
|
|
|
37
|
-
//
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
}
|
|
41
|
-
|
|
42
|
-
//
|
|
43
|
-
//
|
|
44
|
-
//
|
|
45
|
-
//
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
92
|
+
// A surface whose spend is metered (real dollars) vs a subscription surface
|
|
93
|
+
// (metered spend $0; the $ figure is the estimated-equivalent API cost).
|
|
94
|
+
function isMeteredSurface(surface) { return surface === 'api' || surface === 'openai-api'; }
|
|
95
|
+
function isSubscriptionSurface(surface) { return !isMeteredSurface(surface); }
|
|
96
|
+
|
|
97
|
+
// Per-surface retry/timeout policy. A cli/subscription surface spawns a first-party
|
|
98
|
+
// CLI subprocess (cold-start-dominated, occasionally hangs), so it gets a longer
|
|
99
|
+
// per-call timeout (300s) and a longer exponential backoff (baseDelay 30s → 30/60/
|
|
100
|
+
// 120s across the 4 retries); api surfaces keep the tighter 120s / 3s defaults.
|
|
101
|
+
function retryPolicyForSurface(surface) {
|
|
102
|
+
const cli = isSubscriptionSurface(surface);
|
|
103
|
+
return { timeoutMs: cli ? 300000 : 120000, baseDelayMs: cli ? 30000 : 3000, tries: 4 };
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
// The fixed harness preamble Codex prepends to every `codex exec` call. Recorded
|
|
107
|
+
// into openai/cli receipts as run.surface_overhead_note so a reader knows the
|
|
108
|
+
// per-call input-token count is dominated by a constant we do not control.
|
|
109
|
+
const CODEX_OVERHEAD_NOTE =
|
|
110
|
+
'openai/cli surface (codex exec): every call carries a fixed Codex base-instruction '
|
|
111
|
+
+ 'preamble of ~12,000–15,000 input tokens that the harness prepends and we do not '
|
|
112
|
+
+ 'control; the model id is set by us via -m (it is not echoed in the JSONL stream). '
|
|
113
|
+
+ 'Approval prompting is off by default on `codex exec` (no -a flag is passed).';
|
|
114
|
+
|
|
115
|
+
// Send a single-turn prompt and return { text, usage, wall_ms, surface, provider }.
|
|
116
|
+
//
|
|
117
|
+
// usage is the normalized v0.4 record { input_tokens, output_tokens, cached_tokens,
|
|
118
|
+
// wall_ms } on EVERY lane — including the two CLI lanes, which report it in their
|
|
119
|
+
// structured output (`claude -p --output-format json`; the `codex exec --json`
|
|
120
|
+
// JSONL stream). See lib/usage.js for the per-surface shapes and the input-token
|
|
121
|
+
// normalization. Fields the surface does not report stay null, never 0.
|
|
122
|
+
//
|
|
123
|
+
// wall_ms is measured HERE, around the successful attempt, so it means the same
|
|
124
|
+
// thing on all four lanes (retries are excluded — a retried call's latency would
|
|
125
|
+
// describe our backoff, not the model).
|
|
126
|
+
//
|
|
127
|
+
// `temperature` is honoured ONLY on api surfaces; cli surfaces control sampling
|
|
128
|
+
// themselves, so temperature is ignored there and the receipt records that fact.
|
|
129
|
+
async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = null, temperature = undefined }) {
|
|
130
|
+
const surface = surfaceForModel(model);
|
|
131
|
+
const provider = providerForSurface(surface);
|
|
132
|
+
// Offline stub surface: canned completion, zero model calls. The receipt still
|
|
133
|
+
// records the real surface/provider so a stub run is not mistaken for a genuine
|
|
134
|
+
// one at read time — only the generation/judge TEXT is canned.
|
|
135
|
+
if (stubEnabled()) {
|
|
136
|
+
const s = stubComplete({ system, prompt });
|
|
137
|
+
return { ...s, usage: normalizeUsage(s.usage), wall_ms: (s.usage && s.usage.wall_ms) || 0, surface, provider, attempts: 1 };
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
// Per-surface tolerance (see retryPolicyForSurface): a cli/subscription surface
|
|
141
|
+
// spawns a first-party CLI subprocess whose cold-start is slow and occasionally
|
|
142
|
+
// hangs, so it gets a longer per-call timeout (300s) and a longer exponential
|
|
143
|
+
// backoff between the 4 retries (30s / 60s / 120s); api surfaces keep the tighter
|
|
144
|
+
// defaults. An explicit caller timeoutMs still wins.
|
|
145
|
+
const policy = retryPolicyForSurface(surface);
|
|
146
|
+
const effTimeout = timeoutMs != null ? timeoutMs : policy.timeoutMs;
|
|
147
|
+
const baseDelayMs = policy.baseDelayMs;
|
|
148
|
+
|
|
149
|
+
const laneRunner = () => {
|
|
150
|
+
switch (surface) {
|
|
151
|
+
case 'api': return completeAnthropicApi({ system, prompt, model, maxTokens, temperature });
|
|
152
|
+
case 'claude-cli': return completeClaudeCli({ system, prompt, model, timeoutMs: effTimeout });
|
|
153
|
+
case 'openai-api': return completeOpenaiApi({ system, prompt, model, maxTokens, temperature, timeoutMs: effTimeout });
|
|
154
|
+
case 'openai-cli': return completeCodexCli({ system, prompt, model, timeoutMs: effTimeout });
|
|
155
|
+
default: throw new Error(`unknown surface: ${surface}`);
|
|
156
|
+
}
|
|
157
|
+
};
|
|
57
158
|
// A published run makes ~1000s of calls; be patient with transient
|
|
58
|
-
// throttling/cold-starts so one blip doesn't abort a multi-hour grind.
|
|
59
|
-
|
|
60
|
-
|
|
159
|
+
// throttling/cold-starts so one blip doesn't abort a multi-hour grind. `attempts`
|
|
160
|
+
// counts every try (retries included) so the budget can charge for them.
|
|
161
|
+
let attempts = 0;
|
|
162
|
+
// Wall-clock of the attempt that SUCCEEDED (each attempt overwrites, so a
|
|
163
|
+
// retried call reports the latency of the call that actually produced the text,
|
|
164
|
+
// not the accumulated backoff).
|
|
165
|
+
let wallMs = null;
|
|
166
|
+
const runner = () => {
|
|
167
|
+
attempts += 1;
|
|
168
|
+
const t0 = Date.now();
|
|
169
|
+
return withTimeout(laneRunner, effTimeout, `provider(${surface})`)
|
|
170
|
+
.then((r) => { wallMs = Date.now() - t0; return r; });
|
|
171
|
+
};
|
|
172
|
+
try {
|
|
173
|
+
const out = await withRetry(runner, { tries: policy.tries, baseDelayMs });
|
|
174
|
+
const usage = normalizeUsage({ ...(out.usage || {}), wall_ms: wallMs });
|
|
175
|
+
return { ...out, usage, wall_ms: wallMs, surface, provider, attempts };
|
|
176
|
+
} catch (e) {
|
|
177
|
+
// Surface the attempt count so a persistently-failing call can be charged for
|
|
178
|
+
// (and, for a timeout, marked failed_timeout by the runner instead of fatal).
|
|
179
|
+
if (e && typeof e === 'object') e.attempts = attempts;
|
|
180
|
+
throw e;
|
|
181
|
+
}
|
|
61
182
|
}
|
|
62
183
|
|
|
63
|
-
|
|
184
|
+
// ── anthropic/api ─────────────────────────────────────────────────────────────
|
|
185
|
+
async function completeAnthropicApi({ system, prompt, model, maxTokens, temperature }) {
|
|
64
186
|
let Anthropic;
|
|
65
187
|
try { Anthropic = require('@anthropic-ai/sdk'); }
|
|
66
|
-
catch (_e) { throw new Error('
|
|
67
|
-
if (!process.env.ANTHROPIC_API_KEY) throw new Error('
|
|
188
|
+
catch (_e) { throw new Error('the anthropic/api surface requires the @anthropic-ai/sdk package (npm i @anthropic-ai/sdk)'); }
|
|
189
|
+
if (!process.env.ANTHROPIC_API_KEY) throw new Error('the anthropic/api surface requires ANTHROPIC_API_KEY');
|
|
68
190
|
const client = new Anthropic({ apiKey: process.env.ANTHROPIC_API_KEY });
|
|
69
191
|
const params = {
|
|
70
192
|
model: resolveModel(model),
|
|
@@ -75,23 +197,24 @@ async function completeApi({ system, prompt, model, maxTokens, temperature }) {
|
|
|
75
197
|
if (temperature !== undefined) params.temperature = temperature;
|
|
76
198
|
const resp = await client.messages.create(params);
|
|
77
199
|
const text = (resp.content || []).filter((b) => b.type === 'text').map((b) => b.text).join('');
|
|
78
|
-
return {
|
|
79
|
-
text,
|
|
80
|
-
usage: {
|
|
81
|
-
input_tokens: (resp.usage && resp.usage.input_tokens) || 0,
|
|
82
|
-
output_tokens: (resp.usage && resp.usage.output_tokens) || 0,
|
|
83
|
-
},
|
|
84
|
-
};
|
|
200
|
+
return { text, usage: parseAnthropicApiUsage(resp.usage) };
|
|
85
201
|
}
|
|
86
202
|
|
|
87
|
-
|
|
203
|
+
// ── anthropic/cli (claude -p) ──────────────────────────────────────────────────
|
|
204
|
+
function completeClaudeCli({ system, prompt, model, timeoutMs }) {
|
|
88
205
|
return new Promise((resolve, reject) => {
|
|
89
206
|
// Strip ANTHROPIC_API_KEY so the CLI uses the subscription session, not the
|
|
90
207
|
// metered API key. Everything else in the env is preserved.
|
|
91
208
|
const env = { ...process.env };
|
|
92
209
|
delete env.ANTHROPIC_API_KEY;
|
|
93
210
|
|
|
94
|
-
|
|
211
|
+
// v0.4: `--output-format json` returns ONE JSON object carrying both the
|
|
212
|
+
// final text (`result`) and the token usage (`usage`) — the default text
|
|
213
|
+
// output carries no usage at all, which is why usage was previously null on
|
|
214
|
+
// this lane. The text is read from the parsed object; if the CLI ever emits
|
|
215
|
+
// something unparseable we fall back to the raw stdout so a run degrades to
|
|
216
|
+
// the old behaviour (text, no usage) rather than failing.
|
|
217
|
+
const args = ['-p', '--output-format', 'json', '--model', resolveModel(model)];
|
|
95
218
|
if (system) args.push('--append-system-prompt', system);
|
|
96
219
|
|
|
97
220
|
const child = spawn('claude', args, { env, stdio: ['pipe', 'pipe', 'pipe'] });
|
|
@@ -102,17 +225,167 @@ function completeCli({ system, prompt, model, timeoutMs }) {
|
|
|
102
225
|
child.stderr.on('data', (d) => { err += d; });
|
|
103
226
|
child.on('error', (e) => {
|
|
104
227
|
clearTimeout(killer);
|
|
105
|
-
if (e.code === 'ENOENT') reject(new Error("
|
|
228
|
+
if (e.code === 'ENOENT') reject(new Error("the anthropic/cli surface requires the `claude` CLI on PATH. Install it or set CLAUDE_PROVIDER=api."));
|
|
106
229
|
else reject(e);
|
|
107
230
|
});
|
|
108
231
|
child.on('close', (code) => {
|
|
109
232
|
clearTimeout(killer);
|
|
110
233
|
if (code !== 0) return reject(new Error(`claude CLI exited ${code}: ${err.slice(0, 400)}`));
|
|
111
|
-
|
|
234
|
+
const parsed = parseClaudeCliJson(out);
|
|
235
|
+
if (!parsed) return resolve({ text: out.trim(), usage: null });
|
|
236
|
+
if (parsed.isError) return reject(new Error(`claude CLI reported is_error: ${String(parsed.text || '').slice(0, 300)}`));
|
|
237
|
+
resolve({ text: String(parsed.text || '').trim(), usage: parsed.usage });
|
|
112
238
|
});
|
|
113
239
|
child.stdin.write(prompt);
|
|
114
240
|
child.stdin.end();
|
|
115
241
|
});
|
|
116
242
|
}
|
|
117
243
|
|
|
118
|
-
|
|
244
|
+
// ── openai/api (Chat Completions-compatible) ───────────────────────────────────
|
|
245
|
+
// Generic OpenAI-compatible client using global fetch (Node ≥ 18). The base_url
|
|
246
|
+
// is configurable — env OPENAI_BASE_URL wins, else the registry's provider
|
|
247
|
+
// base_url (lib/models.providerConfig), else api.openai.com/v1. Building it
|
|
248
|
+
// base_url-first is deliberate: a future provider is a registry entry with a
|
|
249
|
+
// base_url + api-key env, not a new code path.
|
|
250
|
+
async function completeOpenaiApi({ system, prompt, model, maxTokens, temperature, timeoutMs }) {
|
|
251
|
+
const cfg = openaiProviderConfig();
|
|
252
|
+
const apiKey = process.env[cfg.apiKeyEnv] || process.env.OPENAI_API_KEY;
|
|
253
|
+
if (!apiKey) throw new Error(`the openai/api surface requires ${cfg.apiKeyEnv} (or run the codex CLI surface). Set OPENAI_SURFACE=cli to use the subscription instead.`);
|
|
254
|
+
const baseUrl = (process.env.OPENAI_BASE_URL || cfg.baseUrl || 'https://api.openai.com/v1').replace(/\/+$/, '');
|
|
255
|
+
const messages = [];
|
|
256
|
+
if (system) messages.push({ role: 'system', content: system });
|
|
257
|
+
messages.push({ role: 'user', content: prompt });
|
|
258
|
+
const body = { model: resolveModel(model), messages };
|
|
259
|
+
// Newer OpenAI models use `max_completion_tokens`; classic Chat Completions use
|
|
260
|
+
// `max_tokens`. Send the modern field; harmless on classic-compatible servers
|
|
261
|
+
// that ignore unknown fields, and correct for current OpenAI models.
|
|
262
|
+
if (maxTokens) body.max_completion_tokens = maxTokens;
|
|
263
|
+
if (temperature !== undefined) body.temperature = temperature;
|
|
264
|
+
|
|
265
|
+
const controller = new AbortController();
|
|
266
|
+
const t = setTimeout(() => controller.abort(), timeoutMs);
|
|
267
|
+
let resp;
|
|
268
|
+
try {
|
|
269
|
+
resp = await fetch(`${baseUrl}/chat/completions`, {
|
|
270
|
+
method: 'POST',
|
|
271
|
+
headers: { 'content-type': 'application/json', authorization: `Bearer ${apiKey}` },
|
|
272
|
+
body: JSON.stringify(body),
|
|
273
|
+
signal: controller.signal,
|
|
274
|
+
});
|
|
275
|
+
} finally { clearTimeout(t); }
|
|
276
|
+
const raw = await resp.text();
|
|
277
|
+
if (!resp.ok) throw new Error(`openai/api ${resp.status}: ${raw.slice(0, 300)}`);
|
|
278
|
+
let parsed;
|
|
279
|
+
try { parsed = JSON.parse(raw); } catch (_e) { throw new Error(`openai/api: unparseable response: ${raw.slice(0, 200)}`); }
|
|
280
|
+
const text = ((parsed.choices && parsed.choices[0] && parsed.choices[0].message && parsed.choices[0].message.content) || '');
|
|
281
|
+
return { text: String(text).trim(), usage: parseOpenaiApiUsage(parsed.usage) };
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
// Read the OpenAI provider config (base_url + api-key env) from the registry,
|
|
285
|
+
// lazily (require inside the fn) so provider.js keeps no load-time cycle with
|
|
286
|
+
// lib/models.js. Falls back to the public defaults if the registry omits it.
|
|
287
|
+
function openaiProviderConfig() {
|
|
288
|
+
try {
|
|
289
|
+
const cfg = require('./models').providerConfig('openai');
|
|
290
|
+
if (cfg) return { baseUrl: cfg.base_url, apiKeyEnv: cfg.api_key_env || 'OPENAI_API_KEY' };
|
|
291
|
+
} catch (_e) { /* fall through to defaults */ }
|
|
292
|
+
return { baseUrl: 'https://api.openai.com/v1', apiKeyEnv: 'OPENAI_API_KEY' };
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
// ── openai/cli (codex exec — ChatGPT subscription) ─────────────────────────────
|
|
296
|
+
// Invocation template (from the verified recon, reference_codex_cli):
|
|
297
|
+
// codex exec --json -s read-only --skip-git-repo-check --ephemeral -o <tmp> [-m <model>] "PROMPT"
|
|
298
|
+
// Honoured facts:
|
|
299
|
+
// - `-a/--ask-for-approval` is INVALID on `codex exec` — it is NEVER passed;
|
|
300
|
+
// exec already defaults to approval_policy "never".
|
|
301
|
+
// - the model id is NOT echoed in the JSONL stream; we set -m so the receipt
|
|
302
|
+
// records what WE requested (run.model_id).
|
|
303
|
+
// - auth lives at ~/.codex/auth.json; we do a PRESENCE check only and NEVER
|
|
304
|
+
// read or print its contents.
|
|
305
|
+
// - `codex exec` has no system-prompt flag, so a system prompt (the SKILL.md in
|
|
306
|
+
// with_skill mode) is folded into the prompt text.
|
|
307
|
+
// - stderr emits a harmless "Reading additional input from stdin…" notice.
|
|
308
|
+
const CODEX_EXEC_ARGS = ['exec', '--json', '-s', 'read-only', '--skip-git-repo-check', '--ephemeral'];
|
|
309
|
+
|
|
310
|
+
function codexAuthPresent(homeDir) {
|
|
311
|
+
const home = homeDir || process.env.CODEX_HOME_DIR || os.homedir();
|
|
312
|
+
// CODEX_HOME overrides the auth location if the user set it (codex convention).
|
|
313
|
+
const base = process.env.CODEX_HOME || path.join(home, '.codex');
|
|
314
|
+
return fs.existsSync(path.join(base, 'auth.json'));
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
// Build the exact argv for a codex exec call. Exposed for the gate to assert the
|
|
318
|
+
// template (and that `-a` never appears). `outFile` receives the final message.
|
|
319
|
+
function buildCodexArgs({ model, outFile }) {
|
|
320
|
+
const args = [...CODEX_EXEC_ARGS, '-o', outFile];
|
|
321
|
+
if (model) args.push('-m', resolveModel(model));
|
|
322
|
+
return args;
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
// The FULL argv for a codex call: the template plus a single `-` positional that
|
|
326
|
+
// means "read the prompt from stdin". The prompt is NEVER a positional argv —
|
|
327
|
+
// real SKILL.md files start with `---` (YAML frontmatter) and codex rejects a
|
|
328
|
+
// positional beginning with `--`. Exposed so the gate can lock this in.
|
|
329
|
+
function codexFinalArgs({ model, outFile }) {
|
|
330
|
+
return [...buildCodexArgs({ model, outFile }), '-'];
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
function completeCodexCli({ system, prompt, model, timeoutMs }) {
|
|
334
|
+
return new Promise((resolve, reject) => {
|
|
335
|
+
if (!codexAuthPresent()) {
|
|
336
|
+
return reject(new Error('the openai/cli surface requires codex auth (~/.codex/auth.json). Run `codex login` (or `codex login --device-auth` on a headless box).'));
|
|
337
|
+
}
|
|
338
|
+
// codex exec has no system-prompt flag → fold the system prompt into the
|
|
339
|
+
// prompt text (this is how the SKILL.md reaches the model in with_skill mode).
|
|
340
|
+
const fullPrompt = system ? `${system}\n\n---\n\n${prompt}` : prompt;
|
|
341
|
+
const outFile = path.join(os.tmpdir(), `driftproof-codex-${process.pid}-${Date.now()}-${Math.floor(codexCounter())}.txt`);
|
|
342
|
+
// The prompt is delivered on STDIN, not as a positional argv: real SKILL.md
|
|
343
|
+
// files begin with `---` (YAML frontmatter), and a positional argument that
|
|
344
|
+
// starts with `--` is rejected by codex's arg parser ("unexpected argument
|
|
345
|
+
// '---'"). `-` as the positional tells `codex exec` to read instructions from
|
|
346
|
+
// stdin (per its --help), which is content-agnostic.
|
|
347
|
+
const args = codexFinalArgs({ model, outFile });
|
|
348
|
+
|
|
349
|
+
// v0.4: stdout is CAPTURED (it was 'ignore'). `--json` streams JSONL events
|
|
350
|
+
// there, and the terminal `turn.completed` event carries this call's token
|
|
351
|
+
// usage — the only place codex reports it. The final message still comes from
|
|
352
|
+
// the -o file (cleaner than scraping the stream); the JSONL is read for usage.
|
|
353
|
+
const child = spawn('codex', args, { env: { ...process.env }, stdio: ['pipe', 'pipe', 'pipe'] });
|
|
354
|
+
let err = '';
|
|
355
|
+
let jsonl = '';
|
|
356
|
+
const killer = setTimeout(() => { child.kill('SIGKILL'); }, timeoutMs + 5000);
|
|
357
|
+
child.stdout.on('data', (d) => { jsonl += d; });
|
|
358
|
+
child.stderr.on('data', (d) => { err += d; });
|
|
359
|
+
child.on('error', (e) => {
|
|
360
|
+
clearTimeout(killer);
|
|
361
|
+
if (e.code === 'ENOENT') reject(new Error('the openai/cli surface requires the `codex` CLI on PATH (npm i -g @openai/codex).'));
|
|
362
|
+
else reject(e);
|
|
363
|
+
});
|
|
364
|
+
child.on('close', (code) => {
|
|
365
|
+
clearTimeout(killer);
|
|
366
|
+
let text = '';
|
|
367
|
+
try { text = fs.readFileSync(outFile, 'utf8'); } catch (_e) { /* fall through */ }
|
|
368
|
+
try { fs.unlinkSync(outFile); } catch (_e) { /* best effort */ }
|
|
369
|
+
if (code !== 0) return reject(new Error(`codex exec exited ${code}: ${String(err).slice(0, 400)}`));
|
|
370
|
+
const ev = parseCodexJsonl(jsonl);
|
|
371
|
+
// Prefer the -o file; fall back to the stream's agent_message if it is empty.
|
|
372
|
+
const finalText = String(text || ev.text || '').trim();
|
|
373
|
+
resolve({ text: finalText, usage: ev.usage });
|
|
374
|
+
});
|
|
375
|
+
child.stdin.on('error', () => { /* ignore EPIPE if codex exits before reading all input */ });
|
|
376
|
+
child.stdin.write(fullPrompt);
|
|
377
|
+
child.stdin.end();
|
|
378
|
+
});
|
|
379
|
+
}
|
|
380
|
+
|
|
381
|
+
// A monotonically-increasing counter for temp-file uniqueness that does not use
|
|
382
|
+
// Math.random (kept deterministic-friendly for any harness that forbids it).
|
|
383
|
+
let _codexCounter = 0;
|
|
384
|
+
function codexCounter() { _codexCounter += 1; return _codexCounter; }
|
|
385
|
+
|
|
386
|
+
module.exports = {
|
|
387
|
+
complete, resolveModel, surfaceLabel, providerName, MODEL_ALIASES,
|
|
388
|
+
inferProvider, surfaceForModel, providerForSurface,
|
|
389
|
+
isMeteredSurface, isSubscriptionSurface, retryPolicyForSurface,
|
|
390
|
+
CODEX_OVERHEAD_NOTE, CODEX_EXEC_ARGS, buildCodexArgs, codexFinalArgs, codexAuthPresent,
|
|
391
|
+
};
|
package/lib/receipt.js
CHANGED
|
@@ -13,7 +13,9 @@ const { aggregateBands, combineUncertainty, round } = require('./stats');
|
|
|
13
13
|
const SCHEMA_FILES = {
|
|
14
14
|
'0.1': 'receipt.v0.1.schema.json',
|
|
15
15
|
'0.2': 'receipt.v0.2.schema.json',
|
|
16
|
-
'0.3': 'receipt.schema.json',
|
|
16
|
+
'0.3': 'receipt.v0.3.schema.json',
|
|
17
|
+
'0.3.1': 'receipt.v0.3.1.schema.json',
|
|
18
|
+
'0.4': 'receipt.schema.json',
|
|
17
19
|
};
|
|
18
20
|
|
|
19
21
|
const _validators = {};
|
|
@@ -79,9 +81,13 @@ function aggregate(caseResults) {
|
|
|
79
81
|
// run: { model_id, model_release_date, surface, runner_version, date_utc, judge, registry, transcripts }
|
|
80
82
|
// cases: [ { id, mode, outcome, score, mean, stddev, samples, generation_hash, judge_sample_hashes, threshold, reason, judge } ]
|
|
81
83
|
// editorialReviews: optional [ { url, source, date } ]
|
|
82
|
-
function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED', editorialReviews = null }) {
|
|
83
|
-
|
|
84
|
-
|
|
84
|
+
function buildReceipt({ skill, suite, run, cases, economics = null, verificationLevel = 'TESTED', editorialReviews = null }) {
|
|
85
|
+
// v0.3.1: cases marked failed_timeout are recorded in results.cases but EXCLUDED
|
|
86
|
+
// from aggregates — a band is never fabricated from a case that did not complete.
|
|
87
|
+
const okCases = cases.filter((c) => c.case_status !== 'failed_timeout');
|
|
88
|
+
const failedCount = cases.length - okCases.length;
|
|
89
|
+
const withSkill = okCases.filter((c) => c.mode === 'with_skill');
|
|
90
|
+
const baseline = okCases.filter((c) => c.mode === 'baseline');
|
|
85
91
|
const aggWith = aggregate(withSkill);
|
|
86
92
|
const aggBase = aggregate(baseline);
|
|
87
93
|
|
|
@@ -100,6 +106,9 @@ function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED',
|
|
|
100
106
|
run: {
|
|
101
107
|
model_id: run.model_id,
|
|
102
108
|
model_release_date: run.model_release_date == null ? null : run.model_release_date,
|
|
109
|
+
// v0.3.1: two-axis provider (registry `provider`, else inferred). Defaults
|
|
110
|
+
// to anthropic for any legacy caller that omits it.
|
|
111
|
+
provider: run.provider || 'anthropic',
|
|
103
112
|
surface: run.surface,
|
|
104
113
|
runner_version: run.runner_version,
|
|
105
114
|
date_utc: run.date_utc,
|
|
@@ -124,6 +133,22 @@ function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED',
|
|
|
124
133
|
verification_level: verificationLevel,
|
|
125
134
|
receipt_hash: '',
|
|
126
135
|
};
|
|
136
|
+
// v0.3.1 additive-optional fields (canonicalization sorts keys, so placement
|
|
137
|
+
// here does not affect the hash):
|
|
138
|
+
// run.surface_overhead_note — the fixed harness preamble on the openai/cli surface.
|
|
139
|
+
// skill.tokens — estimated SKILL.md token size (value-per-token axis).
|
|
140
|
+
if (run.surface_overhead_note) receipt.run.surface_overhead_note = run.surface_overhead_note;
|
|
141
|
+
if (skill.tokens != null) receipt.skill.tokens = skill.tokens;
|
|
142
|
+
// v0.4 economics (additive-optional): the frozen prices this receipt's derived
|
|
143
|
+
// dollar figures were computed from, and the derived block itself. A receipt
|
|
144
|
+
// from a surface that reports no usage simply omits both.
|
|
145
|
+
if (run.pricing_snapshot) receipt.run.pricing_snapshot = run.pricing_snapshot;
|
|
146
|
+
if (economics) receipt.economics = economics;
|
|
147
|
+
// v0.3.1: mark the receipt incomplete when any case failed (excluded above).
|
|
148
|
+
if (failedCount > 0) {
|
|
149
|
+
receipt.run.status = 'incomplete';
|
|
150
|
+
receipt.run.failed_case_count = failedCount;
|
|
151
|
+
}
|
|
127
152
|
if (editorialReviews && editorialReviews.length) receipt.editorial_reviews = editorialReviews;
|
|
128
153
|
return sealReceipt(receipt);
|
|
129
154
|
}
|