driftproof 0.11.2 → 0.11.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -11
- package/bin/driftproof +194 -8
- package/config/models.json +3 -2
- package/config.js +2 -2
- package/lib/counts.js +104 -0
- package/lib/decision.js +40 -7
- package/lib/diff.js +12 -3
- package/lib/export.js +4 -1
- package/lib/importers-anthropic.js +451 -0
- package/lib/importers.js +50 -13
- package/lib/receipt.js +72 -3
- package/lib/regrade.js +266 -0
- package/lib/reuse.js +144 -9
- package/lib/run.js +39 -4
- package/lib/skill.js +26 -9
- package/lib/stale.js +194 -0
- package/lib/verdict.js +18 -0
- package/package.json +1 -1
- package/spec/RECEIPT.md +103 -13
- package/spec/receipt.schema.json +330 -15
- package/spec/receipt.v0.7.schema.json +1563 -0
- package/spec/receipt.v0.8.schema.json +1677 -0
- package/spec/stale.v1.schema.json +353 -0
package/lib/importers.js
CHANGED
|
@@ -22,6 +22,7 @@ const { mean, stddev, round, combineUncertainty, aggregateBands } = require('./s
|
|
|
22
22
|
const { outcomeFor } = require('./run');
|
|
23
23
|
const { inferProvider } = require('./provider');
|
|
24
24
|
const { registryStatus } = require('./models');
|
|
25
|
+
const { placeCounts } = require('./counts');
|
|
25
26
|
|
|
26
27
|
const IMPORT_TOOLS = ['agent-skills-eval', 'skillgrade'];
|
|
27
28
|
|
|
@@ -41,10 +42,17 @@ function aggregateMode(cases) {
|
|
|
41
42
|
}
|
|
42
43
|
|
|
43
44
|
// Shared receipt shell for both importers. `judgeBlock` describes the SOURCE
|
|
44
|
-
// tool's grading
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
45
|
+
// tool's grading, never our sampled judge. v0.8 (spec 043): it carries no sample
|
|
46
|
+
// count, because neither source format defines one; `perCase` carries only the counts
|
|
47
|
+
// the source establishes, and placeCounts writes them at the narrowest honest scope.
|
|
48
|
+
// A case with no observation (case_status no_observations) is listed, excluded from
|
|
49
|
+
// every aggregate, and named in excluded_cases with `excludedReasons[id]`.
|
|
50
|
+
function importedReceipt({ tool, skillName, skillVersion, suiteFormat, caseCount, modelId, dateUtc, judgeBlock, cases, comparison, perCase = [], excludedReasons = {} }) {
|
|
51
|
+
const measured = cases.filter((c) => !c.case_status || c.case_status === 'ok');
|
|
52
|
+
const withSkill = measured.filter((c) => c.mode === 'with_skill');
|
|
53
|
+
const baseline = measured.filter((c) => c.mode === 'baseline');
|
|
54
|
+
const excluded = cases.filter((c) => c.case_status && c.case_status !== 'ok')
|
|
55
|
+
.map((c) => ({ id: c.id, modes: [c.mode], reason: excludedReasons[c.id] || 'the source supplied no observation for this case' }));
|
|
48
56
|
const receipt = {
|
|
49
57
|
schema_version: RECEIPT_SCHEMA_VERSION,
|
|
50
58
|
skill: {
|
|
@@ -73,12 +81,13 @@ function importedReceipt({ tool, skillName, skillVersion, suiteFormat, caseCount
|
|
|
73
81
|
},
|
|
74
82
|
results: {
|
|
75
83
|
cases,
|
|
76
|
-
aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline), band_rule: BAND_RULE },
|
|
84
|
+
aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline), band_rule: BAND_RULE, ...(excluded.length ? { excluded_cases: excluded } : {}) },
|
|
77
85
|
},
|
|
78
86
|
comparison,
|
|
79
87
|
verification_level: 'DECLARED',
|
|
80
88
|
receipt_hash: '',
|
|
81
89
|
};
|
|
90
|
+
placeCounts(receipt, perCase);
|
|
82
91
|
return sealReceipt(receipt);
|
|
83
92
|
}
|
|
84
93
|
|
|
@@ -139,7 +148,9 @@ function importAgentSkillsEval(data, { importedAt } = {}) {
|
|
|
139
148
|
caseCount: data.evals.length,
|
|
140
149
|
modelId: data.target || 'unknown',
|
|
141
150
|
dateUtc: data.timestamp || importedAt || new Date().toISOString(),
|
|
142
|
-
|
|
151
|
+
// No count of either kind: the agent-skills-eval benchmark format defines neither
|
|
152
|
+
// (spec 043 AC-5, A-1), so none is written until it does.
|
|
153
|
+
judgeBlock: { temperature: null, sampling: 'external', surface: 'external' },
|
|
143
154
|
cases,
|
|
144
155
|
comparison,
|
|
145
156
|
});
|
|
@@ -158,16 +169,38 @@ function importSkillgrade(data, { importedAt } = {}) {
|
|
|
158
169
|
const graderModel = data.grader_model || 'unknown';
|
|
159
170
|
const defaultThreshold = typeof data.threshold === 'number' ? data.threshold : 0.8;
|
|
160
171
|
const cases = [];
|
|
161
|
-
|
|
172
|
+
const perCase = [];
|
|
173
|
+
const excludedReasons = {};
|
|
162
174
|
for (const task of data.tasks) {
|
|
163
|
-
const
|
|
164
|
-
|
|
165
|
-
|
|
175
|
+
const id = String(task.name);
|
|
176
|
+
// Spec 043 A-3: keep the distinction the source makes. A trials list that is
|
|
177
|
+
// present and empty is zero observations; no trials field is an unknown count.
|
|
178
|
+
// Either way the task is listed and contributes no figure.
|
|
179
|
+
if (!Array.isArray(task.trials)) {
|
|
180
|
+
cases.push({ id, mode: 'with_skill', case_status: 'no_observations' });
|
|
181
|
+
excludedReasons[id] = 'the source carries no trials field for this task: its trial count is unknown';
|
|
182
|
+
continue;
|
|
183
|
+
}
|
|
184
|
+
const rewards = task.trials.map((t) => (typeof t === 'number' ? t : t.reward)).filter((r) => typeof r === 'number');
|
|
185
|
+
if (!task.trials.length) {
|
|
186
|
+
perCase.push({ index: cases.length, counts: { generations_per_arm: 0 } });
|
|
187
|
+
cases.push({ id, mode: 'with_skill', case_status: 'no_observations', samples: [] });
|
|
188
|
+
excludedReasons[id] = 'the source lists no trials for this task (trials is empty): zero observations';
|
|
189
|
+
continue;
|
|
190
|
+
}
|
|
191
|
+
if (!rewards.length) {
|
|
192
|
+
cases.push({ id, mode: 'with_skill', case_status: 'no_observations' });
|
|
193
|
+
excludedReasons[id] = 'the source lists trials for this task but none carries a numeric reward';
|
|
194
|
+
continue;
|
|
195
|
+
}
|
|
196
|
+
// Trials are independent generations (A-2): their number is this case's
|
|
197
|
+
// generation count, and never a judge-sample count.
|
|
198
|
+
perCase.push({ index: cases.length, counts: { generations_per_arm: rewards.length } });
|
|
166
199
|
const m = round(mean(rewards));
|
|
167
200
|
const sd = round(stddev(rewards));
|
|
168
201
|
const threshold = typeof task.threshold === 'number' ? task.threshold : defaultThreshold;
|
|
169
202
|
cases.push({
|
|
170
|
-
id
|
|
203
|
+
id,
|
|
171
204
|
mode: 'with_skill',
|
|
172
205
|
// Same outcome rule as a Driftproof run (borderline when the threshold
|
|
173
206
|
// sits inside mean ± stddev) — a deterministic READING of their numbers.
|
|
@@ -188,11 +221,15 @@ function importSkillgrade(data, { importedAt } = {}) {
|
|
|
188
221
|
// imported verbatim (or the results' model field when present).
|
|
189
222
|
modelId: data.model || data.agent || 'unknown',
|
|
190
223
|
dateUtc: data.timestamp || importedAt || new Date().toISOString(),
|
|
191
|
-
|
|
224
|
+
// No judge-sample count: the skillgrade format defines none, and a trial count is a
|
|
225
|
+
// generation count (spec 043 AC-5, A-1, A-2).
|
|
226
|
+
judgeBlock: { temperature: null, sampling: 'external', surface: 'external' },
|
|
192
227
|
cases,
|
|
228
|
+
perCase,
|
|
229
|
+
excludedReasons,
|
|
193
230
|
// No baseline mode exists in skillgrade — nulls, never a fabricated 0.
|
|
194
231
|
comparison: {
|
|
195
|
-
with_skill_score: round(mean(cases.map((c) => c.mean))),
|
|
232
|
+
with_skill_score: round(mean(cases.filter((c) => !c.case_status).map((c) => c.mean))),
|
|
196
233
|
baseline_score: null,
|
|
197
234
|
delta: null,
|
|
198
235
|
delta_uncertainty: null,
|
package/lib/receipt.js
CHANGED
|
@@ -5,6 +5,7 @@ const fs = require('fs');
|
|
|
5
5
|
const path = require('path');
|
|
6
6
|
const { canonicalize, sha256 } = require('./canonical');
|
|
7
7
|
const { RECEIPT_SCHEMA_VERSION } = require('../config');
|
|
8
|
+
const { placeCounts, judgeCountsDisagree } = require('./counts');
|
|
8
9
|
const { aggregateBands, combineUncertainty, round } = require('./stats');
|
|
9
10
|
|
|
10
11
|
// Schema file per receipt version. The current schema is receipt.schema.json;
|
|
@@ -29,7 +30,15 @@ const SCHEMA_FILES = {
|
|
|
29
30
|
// verdict token a reader derives (UNDERPOWERED), and a v0.6 receipt restamped 0.7
|
|
30
31
|
// is refused by the const, so the number keeps resolving to the schema it meant.
|
|
31
32
|
'0.6': 'receipt.v0.6.schema.json',
|
|
32
|
-
|
|
33
|
+
// v0.7 moved the same way when v0.8 took the current pointer (spec 043): v0.8 makes
|
|
34
|
+
// run.judge.samples optional and adds counts and clocks, and its const refuses a v0.7
|
|
35
|
+
// receipt that has not been restamped, so the number keeps resolving to the schema it meant.
|
|
36
|
+
'0.7': 'receipt.v0.7.schema.json',
|
|
37
|
+
// v0.8 moved the same way when v0.9 took the current pointer (spec 049): v0.9 adds the
|
|
38
|
+
// import record, the harness, the unit and activation, and its const refuses a v0.8 receipt
|
|
39
|
+
// that has not been restamped, so the number keeps resolving to the schema it meant.
|
|
40
|
+
'0.8': 'receipt.v0.8.schema.json',
|
|
41
|
+
'0.9': 'receipt.schema.json',
|
|
33
42
|
};
|
|
34
43
|
|
|
35
44
|
// THE BAND RULE, stated on every receipt (results.aggregates.band_rule, spec
|
|
@@ -82,13 +91,50 @@ function verifyReceiptHash(receipt) {
|
|
|
82
91
|
return receipt.receipt_hash === computeReceiptHash(receipt);
|
|
83
92
|
}
|
|
84
93
|
|
|
94
|
+
// Spec 050: every (id, mode) pair that occurs more than once in results.cases, with
|
|
95
|
+
// the index of each row that carries it. A receipt is one row per case and arm; two
|
|
96
|
+
// rows for one pair are two measurements that every reader keyed by (id, mode) would
|
|
97
|
+
// collapse into one, keeping whichever came last. So a regression measured in the
|
|
98
|
+
// first row could vanish from the verdict while the receipt still validated and its
|
|
99
|
+
// hash still verified (an outside correctness audit, finding 1). This is the one
|
|
100
|
+
// definition of an ambiguous receipt: validation, the verdict, the decision and the
|
|
101
|
+
// CLI all call it, so none of them can disagree about what counts.
|
|
102
|
+
function duplicateCaseRows(receipt) {
|
|
103
|
+
const cases = receipt && receipt.results && Array.isArray(receipt.results.cases) ? receipt.results.cases : [];
|
|
104
|
+
const seen = new Map();
|
|
105
|
+
cases.forEach((c, i) => {
|
|
106
|
+
const key = JSON.stringify([c && c.id, c && c.mode]);
|
|
107
|
+
if (!seen.has(key)) seen.set(key, { id: c && c.id, mode: c && c.mode, rows: [] });
|
|
108
|
+
seen.get(key).rows.push(i);
|
|
109
|
+
});
|
|
110
|
+
return [...seen.values()].filter((d) => d.rows.length > 1);
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
// The sentence every refusal of an ambiguous receipt uses, so the CLI, the verdict
|
|
114
|
+
// and the decision say the same thing about the same receipt.
|
|
115
|
+
function ambiguityLine(dups) {
|
|
116
|
+
return dups.map((d) => `case ${JSON.stringify(d.id)} has ${d.rows.length} ${d.mode} rows (results.cases ${d.rows.join(', ')})`).join('; ');
|
|
117
|
+
}
|
|
118
|
+
|
|
85
119
|
// Validate against the schema matching the receipt's own schema_version (so both
|
|
86
120
|
// v0.1 and v0.2 receipts validate). Returns { valid, errors, version }.
|
|
87
121
|
function validateReceipt(receipt) {
|
|
88
122
|
const version = (receipt && receipt.schema_version) || RECEIPT_SCHEMA_VERSION;
|
|
89
123
|
const validate = getValidator(version);
|
|
90
124
|
const valid = validate(receipt);
|
|
91
|
-
|
|
125
|
+
const errors = valid ? [] : (validate.errors || []);
|
|
126
|
+
// v0.8 (spec 043 AC-2): run.judge.samples and run.counts.judge_samples_per_generation
|
|
127
|
+
// are two spellings of one count; a receipt that gives both must give one value. v0.9
|
|
128
|
+
// keeps both fields, so it keeps the rule.
|
|
129
|
+
if (version === '0.8' || version === '0.9') {
|
|
130
|
+
if (judgeCountsDisagree(receipt)) errors.push({ instancePath: '/run/counts/judge_samples_per_generation', message: 'differs from run.judge.samples' });
|
|
131
|
+
}
|
|
132
|
+
// Spec 050 AC-2: one row per (id, mode), in every schema version. No schema can say
|
|
133
|
+
// it (uniqueness over a pair of fields is outside JSON Schema), so it is said here.
|
|
134
|
+
for (const d of duplicateCaseRows(receipt)) {
|
|
135
|
+
for (const i of d.rows.slice(1)) errors.push({ instancePath: `/results/cases/${i}`, message: `duplicate case id and mode: ${JSON.stringify(d.id)} ${d.mode} is also at /results/cases/${d.rows[0]}` });
|
|
136
|
+
}
|
|
137
|
+
return { valid: errors.length === 0, errors, version };
|
|
92
138
|
}
|
|
93
139
|
|
|
94
140
|
function mean(nums) {
|
|
@@ -224,11 +270,19 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
|
|
|
224
270
|
surface: run.surface,
|
|
225
271
|
runner_version: run.runner_version,
|
|
226
272
|
date_utc: run.date_utc,
|
|
273
|
+
// v0.8 (spec 043 AC-4): when generation began, stamped by the runner before its
|
|
274
|
+
// first generation call; date_utc keeps its meaning.
|
|
275
|
+
...(run.generated_at ? { generated_at: run.generated_at } : {}),
|
|
276
|
+
// v0.8 (spec 043 AC-4; spec 044): a re-judge of frozen outputs. Its archived arms
|
|
277
|
+
// are carried before placeCounts reads them, and the two clocks travel together.
|
|
278
|
+
...(run.arms ? { arms: run.arms } : {}),
|
|
279
|
+
...(run.judged_at ? { judged_at: run.judged_at, grader_revision: run.grader_revision } : {}),
|
|
227
280
|
// v0.3: registry provenance + transcript-retention mode. Defaults keep the
|
|
228
281
|
// honest, cheapest interpretation when a caller omits them.
|
|
229
282
|
registry: run.registry || 'unregistered',
|
|
230
283
|
transcripts: run.transcripts || 'hashes-only',
|
|
231
|
-
|
|
284
|
+
// v0.8: no default judge-sample count; a caller that says nothing leaves it unknown.
|
|
285
|
+
judge: run.judge || { temperature: null, sampling: 'single', surface: run.surface },
|
|
232
286
|
// v0.6 (spec 026 AC-1, AC-2): what answered. Carried from the caller as
|
|
233
287
|
// given and never defaulted: a receipt that does not say what answered it
|
|
234
288
|
// is refused by the schema, which is the point.
|
|
@@ -255,6 +309,8 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
|
|
|
255
309
|
// skill.tokens — estimated SKILL.md token size (value-per-token axis).
|
|
256
310
|
if (run.surface_overhead_note) receipt.run.surface_overhead_note = run.surface_overhead_note;
|
|
257
311
|
if (skill.tokens != null) receipt.skill.tokens = skill.tokens;
|
|
312
|
+
// v0.9, spec 053: the harness the runner read as the run began ({name, version}; absent on a stub run).
|
|
313
|
+
if (run.harness && typeof run.harness.name === 'string') receipt.run.harness = { name: run.harness.name, version: run.harness.version == null ? null : String(run.harness.version) };
|
|
258
314
|
// v0.4 economics (additive-optional): the frozen prices this receipt's derived
|
|
259
315
|
// dollar figures were computed from, and the derived block itself. A receipt
|
|
260
316
|
// from a surface that reports no usage simply omits both.
|
|
@@ -266,9 +322,22 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
|
|
|
266
322
|
receipt.run.failed_case_count = failedCount;
|
|
267
323
|
}
|
|
268
324
|
if (editorialReviews && editorialReviews.length) receipt.editorial_reviews = editorialReviews;
|
|
325
|
+
// v0.8 counts (spec 043 AC-3, AC-7): what the runner establishes, per case. The judge
|
|
326
|
+
// count is the samples the run was asked for; the generation count is the draws the
|
|
327
|
+
// case recorded. A case without a draw set (failed, or a legacy caller) establishes
|
|
328
|
+
// no generation count, so none is written for it.
|
|
329
|
+
const judgeN = receipt.run.judge && Number.isInteger(receipt.run.judge.samples) ? receipt.run.judge.samples : null;
|
|
330
|
+
placeCounts(receipt, cases.map((c, index) => ({
|
|
331
|
+
index,
|
|
332
|
+
counts: {
|
|
333
|
+
...(c.generation && Array.isArray(c.generation.draws) ? { generations_per_arm: c.generation.draws.length } : {}),
|
|
334
|
+
...(judgeN !== null && !caseFailed(c) ? { judge_samples_per_generation: judgeN } : {}),
|
|
335
|
+
},
|
|
336
|
+
})));
|
|
269
337
|
return sealReceipt(receipt);
|
|
270
338
|
}
|
|
271
339
|
|
|
272
340
|
module.exports = {
|
|
273
341
|
buildReceipt, sealReceipt, computeReceiptHash, verifyReceiptHash, validateReceipt, BAND_RULE, comparisonOf, caseFailed, FAILED_STATUSES,
|
|
342
|
+
duplicateCaseRows, ambiguityLine,
|
|
274
343
|
};
|
package/lib/regrade.js
ADDED
|
@@ -0,0 +1,266 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// Regrade: the executor behind lib/reuse.js's `regrade` decision (spec 044).
|
|
5
|
+
//
|
|
6
|
+
// When only the judge (or the grading template) differs between two receipts,
|
|
7
|
+
// triage says the existing generations can be rescored. This does it: it takes a
|
|
8
|
+
// sealed receipt, the skill it was run on and the answers its draws graded, and
|
|
9
|
+
// judges every draw again with the judge it is given. The generation side of the
|
|
10
|
+
// new receipt is the original's, draw for draw; the judge side is new.
|
|
11
|
+
//
|
|
12
|
+
// THE SAME CODE A RUN JUDGES WITH. Each draw goes through lib/run.js judgeCase,
|
|
13
|
+
// its replies through lib/run.js attest, each case through lib/sampling.js
|
|
14
|
+
// acrossDraws, and the receipt through lib/receipt.js buildReceipt. Nothing here
|
|
15
|
+
// grades, aggregates or seals by a second route.
|
|
16
|
+
//
|
|
17
|
+
// THE DRAW SET IS THE ORIGINAL'S. The sampling rule's escalation decides how many
|
|
18
|
+
// generations to draw; with the generations fixed there is nothing for it to
|
|
19
|
+
// decide, so `n_planned` and `stopping_reason` are carried and no draw is added.
|
|
20
|
+
//
|
|
21
|
+
// A DRAW WITH NO ANSWER TO GRADE IS CARRIED, NOT GRADED: one with no
|
|
22
|
+
// generation_hash (the generation timed out) or one cut at the output cap (spec
|
|
23
|
+
// 026 AC-8). A draw whose original judge gave no score has an answer and is graded.
|
|
24
|
+
|
|
25
|
+
const { sha256 } = require('./canonical');
|
|
26
|
+
const { judgeCase, attest, outcomeFor, resolveCallTimeoutMs } = require('./run');
|
|
27
|
+
const { judgeSettings, promptTemplateHash, rubricHash } = require('./judge');
|
|
28
|
+
const { buildReceipt, verifyReceiptHash, FAILED_STATUSES } = require('./receipt');
|
|
29
|
+
const { acrossDraws } = require('./sampling');
|
|
30
|
+
const { resolveModel, surfaceForModel, isMeteredSurface } = require('./provider');
|
|
31
|
+
const { priceForModel, assertRegistered } = require('./models');
|
|
32
|
+
const { perCallCostUSD } = require('./cost');
|
|
33
|
+
const { buildPricingSnapshot, computeEconomics } = require('./value');
|
|
34
|
+
const { canonicalModelId } = require('./usage');
|
|
35
|
+
const { RUNNER_VERSION } = require('../config');
|
|
36
|
+
|
|
37
|
+
const isTimeout = (e) => !!(e && (e.code === 'TIMEOUT' || /tim(e|ed)\s*out/i.test(String((e && e.message) || ''))));
|
|
38
|
+
|
|
39
|
+
// Whether a draw carries an answer the judge can grade.
|
|
40
|
+
function gradable(d) { return !!(d && d.generation_hash && d.truncated !== true); }
|
|
41
|
+
|
|
42
|
+
// Every draw to grade, with its case and mode, in receipt order.
|
|
43
|
+
function drawsToGrade(receipt) {
|
|
44
|
+
const out = [];
|
|
45
|
+
for (const c of receipt.results.cases) for (const d of ((c.generation || {}).draws || [])) if (gradable(d)) out.push({ c, d });
|
|
46
|
+
return out;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
// The answer check (AC-4): every draw to grade has an answer, and each answer is
|
|
50
|
+
// the text its generation_hash was taken over. An answer that fails either is a
|
|
51
|
+
// regrade of something no model wrote.
|
|
52
|
+
function answerProblems(receipt, answers) {
|
|
53
|
+
const bad = [];
|
|
54
|
+
for (const { c, d } of drawsToGrade(receipt)) {
|
|
55
|
+
const where = `${c.id}/${c.mode} draw ${d.draw_index}`;
|
|
56
|
+
const text = answers[d.generation_hash];
|
|
57
|
+
if (typeof text !== 'string') bad.push(`${where}: no answer for generation_hash ${d.generation_hash.slice(0, 12)}`);
|
|
58
|
+
else if (sha256(text) !== d.generation_hash) bad.push(`${where}: the answer's sha256 is not its generation_hash ${d.generation_hash.slice(0, 12)}`);
|
|
59
|
+
}
|
|
60
|
+
return bad;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
// Everything that must hold before the first call, and what the regrade will
|
|
64
|
+
// cost. No call is made. -> { problems, draws, calls, usd }
|
|
65
|
+
function planRegrade({ receipt, skill, answers, judgeModel, samples }) {
|
|
66
|
+
const problems = [];
|
|
67
|
+
if (!verifyReceiptHash(receipt)) problems.push('the receipt_hash does not verify: the receipt was edited after it was sealed');
|
|
68
|
+
if (skill.contentHash !== receipt.skill.content_hash) problems.push(`the skill's content_hash ${String(skill.contentHash).slice(0, 12)} is not the receipt's ${String(receipt.skill.content_hash).slice(0, 12)}`);
|
|
69
|
+
if (skill.suite.suiteHash !== receipt.suite.suite_hash) problems.push(`the suite's hash ${String(skill.suite.suiteHash).slice(0, 12)} is not the receipt's ${String(receipt.suite.suite_hash).slice(0, 12)}`);
|
|
70
|
+
// One row per (case, mode) (spec 044 A-044-4): a receipt with two rows for one case and mode
|
|
71
|
+
// is ambiguous, which spec 050's validation refuses; refused here before any call, not after the
|
|
72
|
+
// spend.
|
|
73
|
+
const rows = new Set();
|
|
74
|
+
for (const c of receipt.results.cases) {
|
|
75
|
+
const key = `${c.id}\u0000${c.mode}`;
|
|
76
|
+
if (rows.has(key)) problems.push(`case ${c.id}/${c.mode} appears in more than one row; the receipt is ambiguous`);
|
|
77
|
+
rows.add(key);
|
|
78
|
+
}
|
|
79
|
+
for (const c of receipt.results.cases) {
|
|
80
|
+
if (!skill.suite.cases.some((k) => k.id === c.id)) problems.push(`case ${c.id} is not in the suite`);
|
|
81
|
+
if (!c.generation || !Array.isArray(c.generation.draws)) problems.push(`case ${c.id}/${c.mode} carries no draw set; a pre-v0.5 receipt cannot be regraded draw by draw`);
|
|
82
|
+
}
|
|
83
|
+
if (!problems.length) problems.push(...answerProblems(receipt, answers));
|
|
84
|
+
const draws = problems.length ? 0 : drawsToGrade(receipt).length;
|
|
85
|
+
const calls = draws * samples;
|
|
86
|
+
const usd = Math.round(calls * perCallCostUSD(resolveModel(judgeModel), 'judge') * 1e4) / 1e4;
|
|
87
|
+
return { problems, draws, calls, usd };
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
// The generation side of a draw, carried as it was.
|
|
91
|
+
function generationSide(d) {
|
|
92
|
+
const g = { stop_reason: d.stop_reason === undefined ? null : d.stop_reason, truncated: d.truncated === true, reported_model: d.reported_model === undefined ? null : d.reported_model };
|
|
93
|
+
return g;
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
// Regrade one sealed receipt. opts: { trusted, budget, timeoutMs, onProgress, nowIso }.
|
|
97
|
+
// -> { receipt, provenance, calls }
|
|
98
|
+
async function regradeReceipt({ receipt, skill, answers, judgeModel, samples, opts = {} }) {
|
|
99
|
+
const judge = resolveModel(judgeModel);
|
|
100
|
+
const modelId = receipt.run.model_id;
|
|
101
|
+
assertRegistered(judge, 'judge model');
|
|
102
|
+
const trusted = !!opts.trusted;
|
|
103
|
+
const budget = opts.budget || null;
|
|
104
|
+
const onProgress = opts.onProgress || (() => {});
|
|
105
|
+
const timeoutMs = resolveCallTimeoutMs(surfaceForModel(judge), opts);
|
|
106
|
+
const judgedAt = new Date().toISOString();
|
|
107
|
+
|
|
108
|
+
let calls = 0;
|
|
109
|
+
const replies = [];
|
|
110
|
+
const cases = [];
|
|
111
|
+
for (const orig of receipt.results.cases) {
|
|
112
|
+
const caseObj = skill.suite.cases.find((k) => k.id === orig.id);
|
|
113
|
+
const mode = orig.mode;
|
|
114
|
+
const draws = [];
|
|
115
|
+
let last = null;
|
|
116
|
+
for (const d of orig.generation.draws) {
|
|
117
|
+
if (!gradable(d)) { draws.push(JSON.parse(JSON.stringify(d))); continue; }
|
|
118
|
+
const gen = generationSide(d);
|
|
119
|
+
const usage = d.usage ? { usage: d.usage } : {};
|
|
120
|
+
onProgress({ case: orig.id, mode, phase: 'judge', draw: d.draw_index, samples });
|
|
121
|
+
let jr;
|
|
122
|
+
try {
|
|
123
|
+
jr = await judgeCase({ caseObj, response: answers[d.generation_hash], generationHash: d.generation_hash, judgeModel: judge, mode, timeoutMs, samples, trusted });
|
|
124
|
+
} catch (e) {
|
|
125
|
+
if (!isTimeout(e)) throw e;
|
|
126
|
+
if (budget) budget.add((e.judgeAttempts || 1) * perCallCostUSD(judge, 'judge'));
|
|
127
|
+
draws.push({ draw_index: d.draw_index, generation_hash: d.generation_hash, status: 'unmeasured', reason: String((e && e.message) || 'timeout').slice(0, 200), samples: [], mean: null, stddev: null, ...gen, ...usage });
|
|
128
|
+
continue;
|
|
129
|
+
}
|
|
130
|
+
calls += samples;
|
|
131
|
+
for (const reply of (jr.replies || [])) {
|
|
132
|
+
if (!reply) continue;
|
|
133
|
+
replies.push(reply);
|
|
134
|
+
attest(reply, judgeModel, 'judge');
|
|
135
|
+
}
|
|
136
|
+
if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judge, 'judge'));
|
|
137
|
+
if (jr.unmeasured) {
|
|
138
|
+
const u = { draw_index: d.draw_index, generation_hash: d.generation_hash, status: 'unmeasured', reason: String(jr.reason || '').slice(0, 200), samples: [], mean: null, stddev: null, ...gen };
|
|
139
|
+
if ((jr.sampleHashes || []).length) u.judge_sample_hashes = jr.sampleHashes;
|
|
140
|
+
Object.assign(u, usage);
|
|
141
|
+
if (jr.judge_usage) u.judge_usage = jr.judge_usage;
|
|
142
|
+
draws.push(u);
|
|
143
|
+
continue;
|
|
144
|
+
}
|
|
145
|
+
const m = {
|
|
146
|
+
draw_index: d.draw_index, generation_hash: d.generation_hash, status: 'measured',
|
|
147
|
+
samples: jr.caseResult.samples, judge_sample_hashes: jr.caseResult.judge_sample_hashes,
|
|
148
|
+
mean: jr.caseResult.mean, stddev: jr.caseResult.stddev, ...gen, ...usage,
|
|
149
|
+
};
|
|
150
|
+
if (jr.caseResult.judge_usage) m.judge_usage = jr.caseResult.judge_usage;
|
|
151
|
+
draws.push(m);
|
|
152
|
+
last = jr;
|
|
153
|
+
}
|
|
154
|
+
const agg = acrossDraws(draws);
|
|
155
|
+
const generation = {
|
|
156
|
+
n_planned: orig.generation.n_planned,
|
|
157
|
+
n_drawn: agg.n_drawn,
|
|
158
|
+
n_measured: agg.n_measured,
|
|
159
|
+
n_unmeasured: agg.n_unmeasured,
|
|
160
|
+
stopping_reason: orig.generation.stopping_reason,
|
|
161
|
+
mean: agg.mean,
|
|
162
|
+
sd: agg.sd,
|
|
163
|
+
judge_sd_mean: agg.judge_sd_mean,
|
|
164
|
+
variance_ratio: agg.variance_ratio,
|
|
165
|
+
variance_ratio_unavailable: agg.variance_ratio_unavailable,
|
|
166
|
+
n_truncated: draws.filter((x) => x.truncated === true).length,
|
|
167
|
+
draws,
|
|
168
|
+
};
|
|
169
|
+
if (!last) {
|
|
170
|
+
const nonTimeout = draws.some((x) => x.status === 'unmeasured' && !/tim(e|ed)\s*out/i.test(String(x.reason || '')));
|
|
171
|
+
const reason = [...draws].reverse().map((x) => x.reason).find(Boolean) || 'timeout';
|
|
172
|
+
cases.push({ id: orig.id, mode, case_status: nonTimeout ? FAILED_STATUSES[1] : FAILED_STATUSES[0], reason, generation });
|
|
173
|
+
continue;
|
|
174
|
+
}
|
|
175
|
+
const caseResult = { ...last.caseResult, generation };
|
|
176
|
+
caseResult.mean = agg.mean;
|
|
177
|
+
caseResult.score = agg.mean;
|
|
178
|
+
caseResult.stddev = agg.sd;
|
|
179
|
+
caseResult.outcome = outcomeFor(agg.mean, agg.sd, caseObj.pass_threshold);
|
|
180
|
+
onProgress({ case: orig.id, mode, phase: 'done', outcome: caseResult.outcome, score: caseResult.mean, stddev: caseResult.stddev });
|
|
181
|
+
cases.push(caseResult);
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
// What answered the judge. A stub judge graded nothing: its receipt is UNVERIFIED
|
|
185
|
+
// with judge surface stub, whatever answered the generations (spec 026 AC-1).
|
|
186
|
+
const judgeStub = replies.length > 0 && replies.every((r) => r.answeredBy === 'stub');
|
|
187
|
+
const judgeModelAnswered = replies.length > 0 && replies.every((r) => r.answeredBy === 'model');
|
|
188
|
+
const judgeBlock = { ...judgeSettings(samples, judge), ...(judgeStub ? { surface: 'stub' } : {}), model_id: judge, prompt_template_hash: promptTemplateHash() };
|
|
189
|
+
const nowIso = opts.nowIso || new Date().toISOString();
|
|
190
|
+
const surface = receipt.run.surface;
|
|
191
|
+
const pricingSnapshot = buildPricingSnapshot({ models: [modelId, judge], lookup: priceForModel, nowIso });
|
|
192
|
+
const economics = computeEconomics({ cases, modelId, judgeModelId: judge, pricingSnapshot, surface, meteredSurface: isMeteredSurface(surface) });
|
|
193
|
+
|
|
194
|
+
// v0.8 (spec 043 AC-4). Every arm was generated in an earlier run, so each is archived:
|
|
195
|
+
// its own model and generation time, carried from the original (the original's own arm
|
|
196
|
+
// override where it has one, else its run's), and its own counts, because an archived arm
|
|
197
|
+
// never inherits run.counts. lib/counts.js placeCounts places counts for arms generated in
|
|
198
|
+
// this run and none is, so they are placed here by its rule: at arm scope when every case
|
|
199
|
+
// of the arm has the same count, else on each case that has one. Nothing is generated in
|
|
200
|
+
// a regrade, so the receipt carries no run.generated_at and no run.counts.
|
|
201
|
+
const arms = {};
|
|
202
|
+
for (const mode of ['with_skill', 'baseline']) {
|
|
203
|
+
const modeCases = cases.filter((c) => c.mode === mode);
|
|
204
|
+
if (!modeCases.length) continue;
|
|
205
|
+
const a = (receipt.run.arms && receipt.run.arms[mode]) || {};
|
|
206
|
+
arms[mode] = { model_id: a.model_id || modelId, generated_at: a.generated_at || receipt.run.generated_at || receipt.run.date_utc };
|
|
207
|
+
const perCase = modeCases.map((c) => ({
|
|
208
|
+
c,
|
|
209
|
+
counts: {
|
|
210
|
+
...(c.generation && Array.isArray(c.generation.draws) ? { generations_per_arm: c.generation.draws.length } : {}),
|
|
211
|
+
...(!c.case_status ? { judge_samples_per_generation: samples } : {}),
|
|
212
|
+
},
|
|
213
|
+
}));
|
|
214
|
+
for (const kind of ['generations_per_arm', 'judge_samples_per_generation']) {
|
|
215
|
+
const values = perCase.map((x) => (Number.isInteger(x.counts[kind]) ? x.counts[kind] : null));
|
|
216
|
+
if (values.every((v) => v !== null) && new Set(values).size === 1) arms[mode].counts = { ...(arms[mode].counts || {}), [kind]: values[0] };
|
|
217
|
+
else for (const x of perCase) if (Number.isInteger(x.counts[kind])) x.c.counts = { ...(x.c.counts || {}), [kind]: x.counts[kind] };
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
const graderRevision = {
|
|
221
|
+
prompt_template_hash: judgeBlock.prompt_template_hash,
|
|
222
|
+
rubric_hashes: cases.map((c) => { const k = skill.suite.cases.find((x) => x.id === c.id); return k ? rubricHash(k.rubric) : null; }),
|
|
223
|
+
};
|
|
224
|
+
const out = buildReceipt({
|
|
225
|
+
skill: { name: receipt.skill.name, version: receipt.skill.version, contentHash: receipt.skill.content_hash, tokens: receipt.skill.tokens },
|
|
226
|
+
suite: { format: receipt.suite.format, suiteHash: receipt.suite.suite_hash, caseCount: receipt.suite.case_count, canary: receipt.suite.canary },
|
|
227
|
+
run: {
|
|
228
|
+
model_id: modelId,
|
|
229
|
+
model_release_date: receipt.run.model_release_date,
|
|
230
|
+
provider: receipt.run.provider,
|
|
231
|
+
surface,
|
|
232
|
+
surface_overhead_note: receipt.run.surface_overhead_note,
|
|
233
|
+
runner_version: RUNNER_VERSION,
|
|
234
|
+
date_utc: nowIso,
|
|
235
|
+
registry: receipt.run.registry,
|
|
236
|
+
transcripts: 'hashes-only',
|
|
237
|
+
judge: judgeBlock,
|
|
238
|
+
pricing_snapshot: pricingSnapshot,
|
|
239
|
+
answered_by: receipt.run.answered_by,
|
|
240
|
+
arms,
|
|
241
|
+
judged_at: judgedAt,
|
|
242
|
+
grader_revision: graderRevision,
|
|
243
|
+
},
|
|
244
|
+
cases,
|
|
245
|
+
economics,
|
|
246
|
+
verificationLevel: receipt.verification_level === 'TESTED' && judgeModelAnswered ? 'TESTED' : 'UNVERIFIED',
|
|
247
|
+
});
|
|
248
|
+
|
|
249
|
+
// Where the regrade came from: what v0.8 has no field for. The judge time, the grader
|
|
250
|
+
// revision and the archived arms are in the receipt (spec 043 AC-4) and are not repeated
|
|
251
|
+
// here; this sidecar is bound to the receipt by its hash.
|
|
252
|
+
const provenance = {
|
|
253
|
+
format: 'driftproof-regrade/2',
|
|
254
|
+
receipt_hash: out.receipt_hash,
|
|
255
|
+
regraded_from: { receipt_hash: receipt.receipt_hash, runner_version: receipt.run.runner_version, date_utc: receipt.run.date_utc, judge: receipt.run.judge },
|
|
256
|
+
generated_at_basis: receipt.run.generated_at || (receipt.run.arms && Object.values(receipt.run.arms).some((x) => x && x.generated_at))
|
|
257
|
+
? "the original receipt's own generated_at"
|
|
258
|
+
: "the original receipt's run.date_utc: a receipt of that schema records no separate generation time",
|
|
259
|
+
draws: { graded: drawsToGrade(receipt).length, carried: receipt.results.cases.reduce((a, c) => a + c.generation.draws.length, 0) - drawsToGrade(receipt).length },
|
|
260
|
+
judge_reported_models: [...new Set(replies.flatMap((r) => (r.reportedModels || []).map((id) => canonicalModelId(id))))].sort(),
|
|
261
|
+
calls,
|
|
262
|
+
};
|
|
263
|
+
return { receipt: out, provenance, calls };
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
module.exports = { regradeReceipt, planRegrade, answerProblems, drawsToGrade, gradable };
|