driftproof 0.8.1 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +87 -16
- package/bin/driftproof +289 -24
- package/config.js +26 -2
- package/lib/checks.js +28 -6
- package/lib/decision.js +425 -0
- package/lib/diff.js +58 -4
- package/lib/importers.js +8 -3
- package/lib/judge.js +69 -12
- package/lib/models.js +52 -11
- package/lib/provider.js +35 -11
- package/lib/receipt.js +51 -11
- package/lib/run.js +221 -20
- package/lib/runner.js +35 -3
- package/lib/skill.js +26 -4
- package/lib/stats.js +12 -3
- package/lib/usage.js +56 -4
- package/lib/value.js +4 -2
- package/lib/verdict.js +6 -1
- package/package.json +3 -2
- package/spec/RECEIPT.md +68 -11
- package/spec/receipt.schema.json +299 -54
- package/spec/receipt.v0.5.schema.json +1318 -0
package/lib/judge.js
CHANGED
|
@@ -58,10 +58,31 @@ function rubricHash(rubric) {
|
|
|
58
58
|
return sha256(JUDGE_SYSTEM + '\n---\n' + String(rubric || '').trim());
|
|
59
59
|
}
|
|
60
60
|
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
61
|
+
// v0.6 (spec 026 AC-11): a digest over the grading TEMPLATE with its three
|
|
62
|
+
// slots empty, recorded once per run as run.judge.prompt_template_hash. The
|
|
63
|
+
// rubric_hash above binds a grade to the case's rubric; this binds every grade
|
|
64
|
+
// of the run to the words around it. A different template is a different judge,
|
|
65
|
+
// and `diff` computes no verdict across one.
|
|
66
|
+
function promptTemplateHash() {
|
|
67
|
+
return sha256(JUDGE_SYSTEM + '\n---\n' + buildJudgePrompt({ task: '', response: '', rubric: '' }));
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
// The score a judge reply carries, or null when it carries none (spec 026
|
|
71
|
+
// AC-3, F2). A non-numeric or non-finite value is not a score; a number outside
|
|
72
|
+
// [0, 1] is on a scale the rubric did not ask for and is not clamped into one
|
|
73
|
+
// (85 is not 1.0). This replaced clamp01, whose silent 0 for a non-finite value
|
|
74
|
+
// scored an absent output as the worst possible one.
|
|
75
|
+
// Stop reasons that mean the surface cut the reply at its output cap: the
|
|
76
|
+
// Messages API's and the claude CLI's `max_tokens`, Chat Completions'
|
|
77
|
+
// `length`. A cut reply is a partial answer (spec 026 AC-8, F4).
|
|
78
|
+
const TRUNCATION_STOP_REASONS = new Set(['max_tokens', 'length']);
|
|
79
|
+
function isTruncated(stopReason) { return TRUNCATION_STOP_REASONS.has(String(stopReason || '')); }
|
|
80
|
+
|
|
81
|
+
function scoreOf(parsed) {
|
|
82
|
+
const x = Number(parsed && parsed.score);
|
|
83
|
+
if (!Number.isFinite(x)) return null;
|
|
84
|
+
if (x < 0 || x > 1) return null;
|
|
85
|
+
return x;
|
|
65
86
|
}
|
|
66
87
|
|
|
67
88
|
// Judge settings for the JUDGE model's surface. Determinism where the surface
|
|
@@ -78,25 +99,50 @@ function judgeSettings(samples, judgeModel) {
|
|
|
78
99
|
return { samples, temperature: null, sampling: 'surface-controlled', surface };
|
|
79
100
|
}
|
|
80
101
|
|
|
81
|
-
// Grade one response once. Returns { score, reason, raw }
|
|
102
|
+
// Grade one response once. Returns { score, reason, raw, ... } for a measured
|
|
103
|
+
// sample, or { unmeasured: true, reason, raw, ... } when the reply carries no
|
|
104
|
+
// score: empty, unparseable, no numeric score, or a score outside [0, 1]. A
|
|
105
|
+
// zero asserts a measurement ("the response satisfied none of the rubric");
|
|
106
|
+
// each of these is the absence of one, and no sample enters any statistic
|
|
107
|
+
// (spec 026 AC-3). The 2026-07 rule that an unparseable judge must never
|
|
108
|
+
// silently pass is kept by the stronger rule: it never silently scores at all.
|
|
82
109
|
// `raw` is the judge's verbatim output text (hashed into the receipt for
|
|
83
110
|
// transcript auditability, and optionally retained under --keep-transcripts).
|
|
84
111
|
async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature, trusted = false }) {
|
|
85
112
|
const prompt = buildJudgePrompt({ task, response, rubric });
|
|
86
|
-
const
|
|
113
|
+
const out = await complete({ system: JUDGE_SYSTEM, prompt, model, maxTokens: 400, timeoutMs, temperature, trusted });
|
|
114
|
+
const { text, attempts, usage } = out;
|
|
115
|
+
const base = { raw: String(text || ''), attempts: attempts || 1, usage, reply: out };
|
|
116
|
+
// A judge reply cut at its output cap carries no complete grade.
|
|
117
|
+
if (isTruncated(out.stopReason)) {
|
|
118
|
+
return { ...base, unmeasured: true, reason: `judge output truncated at the output cap (stop_reason ${out.stopReason})` };
|
|
119
|
+
}
|
|
120
|
+
if (!String(text || '').trim()) {
|
|
121
|
+
return { ...base, unmeasured: true, reason: 'judge output empty (the surface returned no text)' };
|
|
122
|
+
}
|
|
87
123
|
let parsed;
|
|
88
124
|
try {
|
|
89
125
|
parsed = extractJsonObject(text);
|
|
90
126
|
} catch (_e) {
|
|
91
|
-
|
|
92
|
-
// must never silently "pass"), tagged so the caller can see it happened.
|
|
93
|
-
return { score: 0, reason: 'judge output unparseable', unparsed: true, raw: String(text || ''), attempts: attempts || 1, usage };
|
|
127
|
+
return { ...base, unmeasured: true, reason: 'judge output unparseable' };
|
|
94
128
|
}
|
|
95
|
-
|
|
129
|
+
const score = scoreOf(parsed);
|
|
130
|
+
if (score === null) {
|
|
131
|
+
const x = parsed && parsed.score;
|
|
132
|
+
const reason = x === undefined || x === null ? 'judge output carries no numeric score'
|
|
133
|
+
: !Number.isFinite(Number(x)) ? `judge output carries no numeric score (score ${JSON.stringify(x)})`
|
|
134
|
+
: `judge score ${x} outside [0, 1]`;
|
|
135
|
+
return { ...base, unmeasured: true, reason };
|
|
136
|
+
}
|
|
137
|
+
return { ...base, score, reason: String(parsed.reason || '').slice(0, 300) };
|
|
96
138
|
}
|
|
97
139
|
|
|
98
140
|
// Grade a response N times and return the sampled distribution:
|
|
99
141
|
// { samples:[scores], mean, stddev, reason, judge_settings, model_id, rubric_hash }
|
|
142
|
+
// or, when any sample is unmeasured, { unmeasured: true, reason, ... } with NO
|
|
143
|
+
// samples: a partial sample set must never become a band, and a draw one of
|
|
144
|
+
// whose judge samples carried no score is unmeasured as a whole (spec 026
|
|
145
|
+
// AC-3). The remaining samples are not taken.
|
|
100
146
|
// `mean` ± `stddev` is the per-case confidence band (raw spread of the N scores)
|
|
101
147
|
// used by the borderline-outcome rule and per-case drift band-overlap logic.
|
|
102
148
|
// NO DEFAULT TIMEOUT HERE (spec 017 AC-2). This defaulted to 120000, which
|
|
@@ -112,6 +158,7 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
|
|
|
112
158
|
const reasons = [];
|
|
113
159
|
const rawTexts = [];
|
|
114
160
|
const usages = [];
|
|
161
|
+
const replies = []; // v0.6: what answered each judge call, for the runner's attestation
|
|
115
162
|
let attemptsTotal = 0;
|
|
116
163
|
for (let i = 0; i < samples; i++) {
|
|
117
164
|
let r;
|
|
@@ -126,9 +173,18 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
|
|
|
126
173
|
}
|
|
127
174
|
attemptsTotal += r.attempts || 1;
|
|
128
175
|
usages.push(r.usage || null);
|
|
176
|
+
replies.push(r.reply || null);
|
|
177
|
+
rawTexts.push(r.raw || '');
|
|
178
|
+
if (r.unmeasured) {
|
|
179
|
+
return {
|
|
180
|
+
unmeasured: true, reason: r.reason, samples: [], mean: null, stddev: null,
|
|
181
|
+
sample_texts: rawTexts, sample_hashes: rawTexts.map((t) => sha256(t)),
|
|
182
|
+
judge_settings: settings, model_id: model, rubric_hash: rubricHash(rubric),
|
|
183
|
+
attempts: attemptsTotal, usage: sumUsage(usages), replies,
|
|
184
|
+
};
|
|
185
|
+
}
|
|
129
186
|
scores.push(r.score);
|
|
130
187
|
reasons.push(r.reason);
|
|
131
|
-
rawTexts.push(r.raw || '');
|
|
132
188
|
}
|
|
133
189
|
return {
|
|
134
190
|
samples: scores,
|
|
@@ -149,7 +205,8 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
|
|
|
149
205
|
// EXCLUDED from every skill-value figure (lib/value.js): it is a cost we
|
|
150
206
|
// impose to measure, not a cost of running the skill.
|
|
151
207
|
usage: sumUsage(usages),
|
|
208
|
+
replies,
|
|
152
209
|
};
|
|
153
210
|
}
|
|
154
211
|
|
|
155
|
-
module.exports = { gradeOnce, gradeSamples, judgeSettings, rubricHash, buildJudgePrompt, JUDGE_SYSTEM };
|
|
212
|
+
module.exports = { gradeOnce, gradeSamples, judgeSettings, rubricHash, buildJudgePrompt, promptTemplateHash, scoreOf, isTruncated, TRUNCATION_STOP_REASONS, JUDGE_SYSTEM };
|
package/lib/models.js
CHANGED
|
@@ -66,6 +66,26 @@ function resolveRegistry(modelId) {
|
|
|
66
66
|
return { id: canonical, entry, registered: !!entry };
|
|
67
67
|
}
|
|
68
68
|
|
|
69
|
+
// Spec 026 AC-10 (F6): an id that is not a model does not run. After alias
|
|
70
|
+
// resolution the id must have the contract's model-id shape (letters, digits,
|
|
71
|
+
// . _ -) and resolve in the registry; otherwise the run is refused before any
|
|
72
|
+
// call, naming the id and the absolute path of the registry consulted
|
|
73
|
+
// (DRIFTPROOF_REGISTRY when set, else the packaged config/models.json).
|
|
74
|
+
// `registry: "unregistered"` stays a legal receipt value for IMPORTED receipts,
|
|
75
|
+
// whose model field is another tool's word; a run of ours never writes it.
|
|
76
|
+
const MODEL_ID_SHAPE = /^[A-Za-z0-9._-]+$/;
|
|
77
|
+
function assertRegistered(modelId, role = 'model') {
|
|
78
|
+
const given = String(modelId == null ? '' : modelId);
|
|
79
|
+
const { id, registered } = MODEL_ID_SHAPE.test(given) ? resolveRegistry(given) : { id: given, registered: false };
|
|
80
|
+
if (!MODEL_ID_SHAPE.test(given) || !registered) {
|
|
81
|
+
const why = !MODEL_ID_SHAPE.test(given) ? 'is not a model id (letters, digits, . _ - only)' : 'is not in the model registry';
|
|
82
|
+
const e = new Error(`${role} "${given}"${id !== given ? ` (resolved "${id}")` : ''} ${why}: ${REGISTRY_PATH}. An unregistered model does not run; add a registry row (see config/models.json) or point DRIFTPROOF_REGISTRY at a registry that carries it.`);
|
|
83
|
+
e.code = 'UNREGISTERED_MODEL'; e.model = given; e.registry = REGISTRY_PATH;
|
|
84
|
+
throw e;
|
|
85
|
+
}
|
|
86
|
+
return id;
|
|
87
|
+
}
|
|
88
|
+
|
|
69
89
|
// The value stamped into receipt.run.registry.
|
|
70
90
|
function registryStatus(modelId) {
|
|
71
91
|
return resolveRegistry(modelId).registered ? 'registered' : 'unregistered';
|
|
@@ -140,23 +160,43 @@ function familyPredecessor(modelId) {
|
|
|
140
160
|
return chooseFrom[0].id;
|
|
141
161
|
}
|
|
142
162
|
|
|
143
|
-
//
|
|
144
|
-
//
|
|
145
|
-
//
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
163
|
+
// THE PRICE OF A MODEL NOBODY HAS PRICED YET.
|
|
164
|
+
//
|
|
165
|
+
// READ THIS BEFORE TRUSTING A NUMBER IT RETURNS. Nothing here reads a price
|
|
166
|
+
// SOURCE. The family and the tier are inferred from the id, and the rate is
|
|
167
|
+
// looked up in the table below, so the result is a GUESS whose only input is the
|
|
168
|
+
// string. It is extracted into its own function, and named, because the guess
|
|
169
|
+
// was being made silently inside a function whose job looked like registration.
|
|
170
|
+
//
|
|
171
|
+
// KNOWN DEFECT, spec 021 F-W2, deliberately NOT fixed here. The Anthropic branch
|
|
172
|
+
// falls through to the OPUS rate of 5/25 for every family that is not `haiku` or
|
|
173
|
+
// `sonnet`. Fable's published rate is 10/50, exactly twice opus, which is the
|
|
174
|
+
// whole of the "the watcher read exactly half the snapshot on both fields"
|
|
175
|
+
// observation of 2026-09-03: it is not a halving and not a misread, it is a
|
|
176
|
+
// wrong-tier fallback. `DEFAULT_PRICE` above IS the conservative upper bound the
|
|
177
|
+
// comment on this function used to claim, and it is not consulted. Recorded on
|
|
178
|
+
// the spec 021 carry list, tagged 026, with the reason it is left alone.
|
|
179
|
+
//
|
|
180
|
+
// scripts/release-watch.js no longer registers anything, and records what this
|
|
181
|
+
// returns as `price_verified: false` beside a source naming this function.
|
|
182
|
+
function inferredPrice(id) {
|
|
149
183
|
const family = familyOf(id);
|
|
150
184
|
const provider = inferProviderFromId(id);
|
|
151
|
-
// Best-effort tier from family; unknown families default to frontier pricing.
|
|
152
|
-
// A newly-discovered id gets a CONSERVATIVE (upper-bound) price so the budget
|
|
153
|
-
// guard never under-projects an unknown model — exact rates are set by hand
|
|
154
|
-
// when the model is reviewed for a published run.
|
|
155
185
|
const tier = /haiku|luna|mini|nano/.test(family) || /luna|mini|nano/.test(String(id)) ? 'cheap'
|
|
156
186
|
: /sonnet|terra/.test(family) ? 'standard' : 'frontier';
|
|
157
187
|
const price = provider === 'openai'
|
|
158
188
|
? (tier === 'cheap' ? { input: 1.0, output: 6.0 } : tier === 'standard' ? { input: 2.5, output: 15.0 } : { input: 5.0, output: 30.0 })
|
|
159
189
|
: (family === 'haiku' ? { input: 1.0, output: 5.0 } : family === 'sonnet' ? { input: 3.0, output: 15.0 } : { input: 5.0, output: 25.0 });
|
|
190
|
+
return { family, provider, tier, price };
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
// Append a newly-discovered model id to the registry file (release trigger).
|
|
194
|
+
// `firstSeen` is the first-seen date recorded as `released`. Idempotent: a no-op
|
|
195
|
+
// if the id already exists. Returns the added entry (or null if already present).
|
|
196
|
+
function addDiscoveredModel(id, { firstSeen = null, registryPath = REGISTRY_PATH } = {}) {
|
|
197
|
+
const raw = JSON.parse(fs.readFileSync(registryPath, 'utf8'));
|
|
198
|
+
if (raw.models.some((m) => m.id === id)) return null;
|
|
199
|
+
const { family, provider, tier, price } = inferredPrice(id);
|
|
160
200
|
const entry = {
|
|
161
201
|
id, family, provider, released: firstSeen,
|
|
162
202
|
input_price: price.input, output_price: price.output, tier,
|
|
@@ -171,5 +211,6 @@ function addDiscoveredModel(id, { firstSeen = null, registryPath = REGISTRY_PATH
|
|
|
171
211
|
module.exports = {
|
|
172
212
|
loadRegistry, resolveRegistry, registryStatus, priceForModel, isJudgeEligible,
|
|
173
213
|
assertJudgeEligible, providerForModel, providerConfig, inferProviderFromId,
|
|
174
|
-
familyOf, familyPredecessor, addDiscoveredModel, DEFAULT_PRICE, REGISTRY_PATH,
|
|
214
|
+
familyOf, familyPredecessor, addDiscoveredModel, inferredPrice, DEFAULT_PRICE, REGISTRY_PATH,
|
|
215
|
+
assertRegistered, MODEL_ID_SHAPE,
|
|
175
216
|
};
|
package/lib/provider.js
CHANGED
|
@@ -9,6 +9,7 @@ const path = require('path');
|
|
|
9
9
|
const cp = require('child_process');
|
|
10
10
|
const { withRetry, withTimeout } = require('./json');
|
|
11
11
|
const { stubComplete, stubEnabled } = require('./stub');
|
|
12
|
+
const { readAnthropicApiResponse, readOpenaiApiResponse } = require('./usage');
|
|
12
13
|
const {
|
|
13
14
|
parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage, normalizeUsage,
|
|
14
15
|
} = require('./usage');
|
|
@@ -134,12 +135,17 @@ const CODEX_OVERHEAD_NOTE =
|
|
|
134
135
|
async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = null, temperature = undefined, trusted = false }) {
|
|
135
136
|
const surface = surfaceForModel(model);
|
|
136
137
|
const provider = providerForSurface(surface);
|
|
137
|
-
// Offline stub surface: canned completion, zero model calls. The
|
|
138
|
-
//
|
|
139
|
-
//
|
|
138
|
+
// Offline stub surface: canned completion, zero model calls. The reply says
|
|
139
|
+
// so: surface `stub`, answeredBy `stub`, no model reported, no stop reason,
|
|
140
|
+
// no isolation (nothing was spawned). Spec 026 F1: the receipt used to record
|
|
141
|
+
// the REAL surface name here and read TESTED while the text was canned; now
|
|
142
|
+
// the receipt's surface, level and answered_by are derived from what this
|
|
143
|
+
// function returned, and a stub run reads UNVERIFIED with surface stub. The
|
|
144
|
+
// requested model's provider is still reported, so a reader knows what was
|
|
145
|
+
// asked for.
|
|
140
146
|
if (stubEnabled()) {
|
|
141
147
|
const s = stubComplete({ system, prompt });
|
|
142
|
-
return { ...s, usage: normalizeUsage(s.usage), wall_ms: (s.usage && s.usage.wall_ms) || 0, surface, provider, attempts: 1 };
|
|
148
|
+
return { ...s, usage: normalizeUsage(s.usage), wall_ms: (s.usage && s.usage.wall_ms) || 0, surface: 'stub', provider, attempts: 1, answeredBy: 'stub', reportedModels: null, stopReason: null, isolation: 'none' };
|
|
143
149
|
}
|
|
144
150
|
|
|
145
151
|
// Per-surface tolerance (see retryPolicyForSurface): a cli/subscription surface
|
|
@@ -177,7 +183,16 @@ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = n
|
|
|
177
183
|
try {
|
|
178
184
|
const out = await withRetry(runner, { tries: policy.tries, baseDelayMs });
|
|
179
185
|
const usage = normalizeUsage({ ...(out.usage || {}), wall_ms: wallMs });
|
|
180
|
-
|
|
186
|
+
// v0.6 (spec 026 AC-1, AC-2, AC-8): a model surface answered; what it said
|
|
187
|
+
// served the call, why the reply stopped, and which spawn path was taken
|
|
188
|
+
// are what the lane could read, null where its surface reports nothing.
|
|
189
|
+
return {
|
|
190
|
+
...out, usage, wall_ms: wallMs, surface, provider, attempts,
|
|
191
|
+
answeredBy: 'model',
|
|
192
|
+
reportedModels: Array.isArray(out.reportedModels) && out.reportedModels.length ? out.reportedModels : null,
|
|
193
|
+
stopReason: typeof out.stopReason === 'string' && out.stopReason ? out.stopReason : null,
|
|
194
|
+
isolation: out.isolation || 'none',
|
|
195
|
+
};
|
|
181
196
|
} catch (e) {
|
|
182
197
|
// Surface the attempt count so a persistently-failing call can be charged for
|
|
183
198
|
// (and, for a timeout, marked failed_timeout by the runner instead of fatal).
|
|
@@ -202,7 +217,8 @@ async function completeAnthropicApi({ system, prompt, model, maxTokens, temperat
|
|
|
202
217
|
if (temperature !== undefined) params.temperature = temperature;
|
|
203
218
|
const resp = await client.messages.create(params);
|
|
204
219
|
const text = (resp.content || []).filter((b) => b.type === 'text').map((b) => b.text).join('');
|
|
205
|
-
|
|
220
|
+
// v0.6: the response names the model that served it and why it stopped.
|
|
221
|
+
return { text, usage: parseAnthropicApiUsage(resp.usage), ...readAnthropicApiResponse(resp), isolation: 'none' };
|
|
206
222
|
}
|
|
207
223
|
|
|
208
224
|
// ── isolation: the eval-user hop (spec 022) ───────────────────────────────────
|
|
@@ -275,6 +291,10 @@ function isolatedEnv(user) {
|
|
|
275
291
|
return { HOME: home, PATH: `${home}/.local/bin:/usr/bin:/bin` };
|
|
276
292
|
}
|
|
277
293
|
|
|
294
|
+
// The isolation a plan records into the receipt (run.answered_by.isolation,
|
|
295
|
+
// spec 026 AC-2): the hop is `eval-user`, the legacy spawn `same-user`.
|
|
296
|
+
function isolationOf(plan) { return plan && plan.mode === 'isolated' ? 'eval-user' : 'same-user'; }
|
|
297
|
+
|
|
278
298
|
// The spawn plan for one CLI call: { mode, user, file, args, env }. Pure, so the
|
|
279
299
|
// gate can inspect what WOULD be spawned. Untrusted → the hop; trusted → the
|
|
280
300
|
// legacy same-user spawn exactly as it was (env inherited; the claude lane still
|
|
@@ -376,9 +396,10 @@ async function completeClaudeCli({ system, prompt, model, timeoutMs, trusted = f
|
|
|
376
396
|
throw new Error(hop || `claude CLI exited ${r.code}: ${r.stderr.slice(0, 400)}`);
|
|
377
397
|
}
|
|
378
398
|
const parsed = parseClaudeCliJson(r.stdout);
|
|
379
|
-
if (!parsed) return { text: r.stdout.trim(), usage: null };
|
|
399
|
+
if (!parsed) return { text: r.stdout.trim(), usage: null, stopReason: null, reportedModels: null, isolation: isolationOf(plan) };
|
|
380
400
|
if (parsed.isError) throw new Error(`claude CLI reported is_error: ${String(parsed.text || '').slice(0, 300)}`);
|
|
381
|
-
|
|
401
|
+
// v0.6: the CLI's own stop_reason and the modelUsage map's canonical ids.
|
|
402
|
+
return { text: String(parsed.text || '').trim(), usage: parsed.usage, stopReason: parsed.stopReason, reportedModels: parsed.reportedModels, isolation: isolationOf(plan) };
|
|
382
403
|
}
|
|
383
404
|
|
|
384
405
|
// ── openai/api (Chat Completions-compatible) ───────────────────────────────────
|
|
@@ -418,7 +439,8 @@ async function completeOpenaiApi({ system, prompt, model, maxTokens, temperature
|
|
|
418
439
|
let parsed;
|
|
419
440
|
try { parsed = JSON.parse(raw); } catch (_e) { throw new Error(`openai/api: unparseable response: ${raw.slice(0, 200)}`); }
|
|
420
441
|
const text = ((parsed.choices && parsed.choices[0] && parsed.choices[0].message && parsed.choices[0].message.content) || '');
|
|
421
|
-
|
|
442
|
+
// v0.6: the response's model and the first choice's finish_reason.
|
|
443
|
+
return { text: String(text).trim(), usage: parseOpenaiApiUsage(parsed.usage), ...readOpenaiApiResponse(parsed), isolation: 'none' };
|
|
422
444
|
}
|
|
423
445
|
|
|
424
446
|
// Read the OpenAI provider config (base_url + api-key env) from the registry,
|
|
@@ -515,7 +537,9 @@ async function completeCodexCli({ system, prompt, model, timeoutMs, trusted = fa
|
|
|
515
537
|
throw new Error(hop || `codex exec exited ${r.code}: ${String(r.stderr).slice(0, 400)}`);
|
|
516
538
|
}
|
|
517
539
|
const ev = parseCodexJsonl(r.stdout);
|
|
518
|
-
|
|
540
|
+
// codex reports neither the serving model nor a stop reason on its stream;
|
|
541
|
+
// both read null and the receipt says the surface did not report them.
|
|
542
|
+
return { text: String(ev.text || '').trim(), usage: ev.usage, stopReason: ev.stopReason, reportedModels: ev.reportedModels, isolation: isolationOf(plan) };
|
|
519
543
|
}
|
|
520
544
|
|
|
521
545
|
module.exports = {
|
|
@@ -525,5 +549,5 @@ module.exports = {
|
|
|
525
549
|
CODEX_OVERHEAD_NOTE, CODEX_EXEC_ARGS, buildCodexArgs, codexFinalArgs, codexAuthPresent,
|
|
526
550
|
// spec 022 — the eval-user hop
|
|
527
551
|
SUDO_BIN, EVAL_USER_DEFAULT, ISOLATED_ENV_ALLOWLIST, ISOLATED_WRAPPER, HOP_LABEL,
|
|
528
|
-
evalUser, isolatedEnv, buildSpawnPlan, runIsolated,
|
|
552
|
+
evalUser, isolatedEnv, buildSpawnPlan, runIsolated, isolationOf,
|
|
529
553
|
};
|
package/lib/receipt.js
CHANGED
|
@@ -20,9 +20,30 @@ const SCHEMA_FILES = {
|
|
|
20
20
|
// published v0.4 receipt asserts conformance by NUMBER, and the number has to
|
|
21
21
|
// keep resolving to the schema it meant (AC-10).
|
|
22
22
|
'0.4': 'receipt.v0.4.schema.json',
|
|
23
|
-
|
|
23
|
+
// v0.5 moved the same way when v0.6 took the current pointer (spec 026): the
|
|
24
|
+
// six report-008 receipts assert v0.5 by number, and v0.6 is not additive for
|
|
25
|
+
// the validator (it requires the receipt to say what answered it), so the
|
|
26
|
+
// number has to keep resolving to the schema it meant.
|
|
27
|
+
'0.5': 'receipt.v0.5.schema.json',
|
|
28
|
+
'0.6': 'receipt.schema.json',
|
|
24
29
|
};
|
|
25
30
|
|
|
31
|
+
// THE BAND RULE, stated on every receipt (results.aggregates.band_rule, spec
|
|
32
|
+
// 026 AC-6) so a reader recomputes the summary from the rows by the formula
|
|
33
|
+
// it names rather than by guessing which of the two "bands" a receipt carries.
|
|
34
|
+
// Per case, stddev is the spread of judge samples; per arm, it is this.
|
|
35
|
+
const BAND_RULE = 'per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases';
|
|
36
|
+
|
|
37
|
+
// THE ONE PREDICATE for "this case did not complete" (spec 026 AC-3). Every
|
|
38
|
+
// reader in lib/ and scripts/ asks this, never the failed_timeout literal: a
|
|
39
|
+
// reader that excluded by that literal admitted a failed_unmeasured case (an
|
|
40
|
+
// empty generation, a judge with no score) into its statistics as if it had
|
|
41
|
+
// been measured. A case is failed when its case_status is present and is not
|
|
42
|
+
// 'ok'; the status names what was observed, and both failure statuses are
|
|
43
|
+
// recorded without fabricated samples or hashes.
|
|
44
|
+
const FAILED_STATUSES = ['failed_timeout', 'failed_unmeasured'];
|
|
45
|
+
function caseFailed(c) { return !!(c && c.case_status && c.case_status !== 'ok'); }
|
|
46
|
+
|
|
26
47
|
const _validators = {};
|
|
27
48
|
// Lazily compile the JSON Schema validator (ajv) for a given version. Kept lazy
|
|
28
49
|
// so the library can be required without ajv present (pure hashing utilities).
|
|
@@ -80,6 +101,27 @@ function aggregate(caseResults) {
|
|
|
80
101
|
return { case_count: caseResults.length, pass_count: passes, borderline_count: borderline, mean_score: band.mean, stddev: band.stddev };
|
|
81
102
|
}
|
|
82
103
|
|
|
104
|
+
// The comparison block from the two arms' aggregates. Every null it carries is
|
|
105
|
+
// named: `delta_uncertainty_unavailable` is `no_cases` when an arm has no
|
|
106
|
+
// included case (then every score is null too: the mean of nothing is not a
|
|
107
|
+
// number) and `single_case` when an arm has one (a mean, no band). Otherwise
|
|
108
|
+
// the block is numeric and the field is absent (spec 026 AC-7).
|
|
109
|
+
function comparisonOf(aggWith, aggBase) {
|
|
110
|
+
const noCases = aggWith.case_count === 0 || aggBase.case_count === 0;
|
|
111
|
+
const single = !noCases && (aggWith.case_count < 2 || aggBase.case_count < 2);
|
|
112
|
+
const cmp = {
|
|
113
|
+
with_skill_score: aggWith.mean_score,
|
|
114
|
+
baseline_score: aggBase.mean_score,
|
|
115
|
+
delta: noCases ? null : round(aggWith.mean_score - aggBase.mean_score),
|
|
116
|
+
// Combined uncertainty of the delta: quadrature sum of the two aggregate
|
|
117
|
+
// bands. Diff uses this for the headline "within noise" vs real-move rule.
|
|
118
|
+
delta_uncertainty: noCases ? null : combineUncertainty(aggWith.stddev, aggBase.stddev),
|
|
119
|
+
};
|
|
120
|
+
if (noCases) cmp.delta_uncertainty_unavailable = 'no_cases';
|
|
121
|
+
else if (single) cmp.delta_uncertainty_unavailable = 'single_case';
|
|
122
|
+
return cmp;
|
|
123
|
+
}
|
|
124
|
+
|
|
83
125
|
// Assemble a full receipt from the runner's raw pieces, seal it, and return it.
|
|
84
126
|
// skill: { name, version, contentHash }
|
|
85
127
|
// suite: { format, suiteHash, caseCount }
|
|
@@ -109,7 +151,7 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
|
|
|
109
151
|
// The excluded case STAYS in `results.cases`. It is removed from the mean, not
|
|
110
152
|
// from the record — deleting the evidence of a failure is a different and worse
|
|
111
153
|
// defect than averaging over it.
|
|
112
|
-
const armUnusable = (c) => c
|
|
154
|
+
const armUnusable = (c) => caseFailed(c)
|
|
113
155
|
|| (c.mean == null && c.score == null);
|
|
114
156
|
const excludedIds = new Map();
|
|
115
157
|
for (const c of cases) {
|
|
@@ -183,25 +225,23 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
|
|
|
183
225
|
registry: run.registry || 'unregistered',
|
|
184
226
|
transcripts: run.transcripts || 'hashes-only',
|
|
185
227
|
judge: run.judge || { samples: 1, temperature: null, sampling: 'single', surface: run.surface },
|
|
228
|
+
// v0.6 (spec 026 AC-1, AC-2): what answered. Carried from the caller as
|
|
229
|
+
// given and never defaulted: a receipt that does not say what answered it
|
|
230
|
+
// is refused by the schema, which is the point.
|
|
231
|
+
...(run.answered_by ? { answered_by: run.answered_by } : {}),
|
|
186
232
|
},
|
|
187
233
|
results: {
|
|
188
234
|
cases,
|
|
189
235
|
aggregates: {
|
|
190
236
|
with_skill: aggWith,
|
|
191
237
|
baseline: aggBase,
|
|
238
|
+
band_rule: BAND_RULE,
|
|
192
239
|
// Present only when something was excluded, so a clean run's receipt is
|
|
193
240
|
// unchanged and the archive does not acquire an empty field.
|
|
194
241
|
...(excludedCases.length ? { excluded_cases: excludedCases } : {}),
|
|
195
242
|
},
|
|
196
243
|
},
|
|
197
|
-
comparison:
|
|
198
|
-
with_skill_score: aggWith.mean_score,
|
|
199
|
-
baseline_score: aggBase.mean_score,
|
|
200
|
-
delta: round(aggWith.mean_score - aggBase.mean_score),
|
|
201
|
-
// Combined uncertainty of the delta: quadrature sum of the two aggregate
|
|
202
|
-
// bands. Diff uses this for the headline "within noise" vs real-move rule.
|
|
203
|
-
delta_uncertainty: combineUncertainty(aggWith.stddev, aggBase.stddev),
|
|
204
|
-
},
|
|
244
|
+
comparison: comparisonOf(aggWith, aggBase),
|
|
205
245
|
verification_level: verificationLevel,
|
|
206
246
|
receipt_hash: '',
|
|
207
247
|
};
|
|
@@ -226,5 +266,5 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
|
|
|
226
266
|
}
|
|
227
267
|
|
|
228
268
|
module.exports = {
|
|
229
|
-
buildReceipt, sealReceipt, computeReceiptHash, verifyReceiptHash, validateReceipt,
|
|
269
|
+
buildReceipt, sealReceipt, computeReceiptHash, verifyReceiptHash, validateReceipt, BAND_RULE, comparisonOf, caseFailed, FAILED_STATUSES,
|
|
230
270
|
};
|