driftproof 0.11.2 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/receipt.js CHANGED
@@ -5,6 +5,7 @@ const fs = require('fs');
5
5
  const path = require('path');
6
6
  const { canonicalize, sha256 } = require('./canonical');
7
7
  const { RECEIPT_SCHEMA_VERSION } = require('../config');
8
+ const { placeCounts, judgeCountsDisagree } = require('./counts');
8
9
  const { aggregateBands, combineUncertainty, round } = require('./stats');
9
10
 
10
11
  // Schema file per receipt version. The current schema is receipt.schema.json;
@@ -29,7 +30,15 @@ const SCHEMA_FILES = {
29
30
  // verdict token a reader derives (UNDERPOWERED), and a v0.6 receipt restamped 0.7
30
31
  // is refused by the const, so the number keeps resolving to the schema it meant.
31
32
  '0.6': 'receipt.v0.6.schema.json',
32
- '0.7': 'receipt.schema.json',
33
+ // v0.7 moved the same way when v0.8 took the current pointer (spec 043): v0.8 makes
34
+ // run.judge.samples optional and adds counts and clocks, and its const refuses a v0.7
35
+ // receipt that has not been restamped, so the number keeps resolving to the schema it meant.
36
+ '0.7': 'receipt.v0.7.schema.json',
37
+ // v0.8 moved the same way when v0.9 took the current pointer (spec 049): v0.9 adds the
38
+ // import record, the harness, the unit and activation, and its const refuses a v0.8 receipt
39
+ // that has not been restamped, so the number keeps resolving to the schema it meant.
40
+ '0.8': 'receipt.v0.8.schema.json',
41
+ '0.9': 'receipt.schema.json',
33
42
  };
34
43
 
35
44
  // THE BAND RULE, stated on every receipt (results.aggregates.band_rule, spec
@@ -48,20 +57,28 @@ const BAND_RULE = 'per arm: the sample standard deviation (n-1) of the per-case
48
57
  const FAILED_STATUSES = ['failed_timeout', 'failed_unmeasured'];
49
58
  function caseFailed(c) { return !!(c && c.case_status && c.case_status !== 'ok'); }
50
59
 
51
- const _validators = {};
60
+ // Spec 062 (register row 2): the table is read by OWN key and the cache is a Map. Until this
61
+ // the lookup was `SCHEMA_FILES[version]`, so a receipt whose schema_version named an
62
+ // Object.prototype property ("constructor", "toString") found something truthy, fell back to
63
+ // the current schema's cached validator and read VALID; and a version nobody knew was checked
64
+ // against the current schema as though it had claimed it. A version that is not a row here has
65
+ // no schema, and a receipt with no schema is not valid.
66
+ const _validators = new Map();
52
67
  // Lazily compile the JSON Schema validator (ajv) for a given version. Kept lazy
53
68
  // so the library can be required without ajv present (pure hashing utilities).
69
+ // Returns null for a version the table does not hold.
54
70
  function getValidator(version) {
55
- const v = SCHEMA_FILES[version] ? version : RECEIPT_SCHEMA_VERSION;
56
- if (_validators[v]) return _validators[v];
71
+ if (typeof version !== 'string' || !Object.hasOwn(SCHEMA_FILES, version)) return null;
72
+ if (_validators.has(version)) return _validators.get(version);
57
73
  let Ajv;
58
74
  // The schema is JSON Schema draft 2020-12, so use ajv's 2020 build.
59
75
  try { Ajv = require('ajv/dist/2020'); }
60
76
  catch (_e) { throw new Error('receipt validation requires the `ajv` package (npm install)'); }
61
- const schema = JSON.parse(fs.readFileSync(path.join(__dirname, '..', 'spec', SCHEMA_FILES[v]), 'utf8'));
77
+ const schema = JSON.parse(fs.readFileSync(path.join(__dirname, '..', 'spec', SCHEMA_FILES[version]), 'utf8'));
62
78
  const ajv = new Ajv({ allErrors: true, strict: false });
63
- _validators[v] = ajv.compile(schema);
64
- return _validators[v];
79
+ const validate = ajv.compile(schema);
80
+ _validators.set(version, validate);
81
+ return validate;
65
82
  }
66
83
 
67
84
  // Compute the receipt_hash: sha256 over the canonical receipt JSON with the
@@ -82,13 +99,145 @@ function verifyReceiptHash(receipt) {
82
99
  return receipt.receipt_hash === computeReceiptHash(receipt);
83
100
  }
84
101
 
102
+ // Spec 050: every (id, mode) pair that occurs more than once in results.cases, with
103
+ // the index of each row that carries it. A receipt is one row per case and arm; two
104
+ // rows for one pair are two measurements that every reader keyed by (id, mode) would
105
+ // collapse into one, keeping whichever came last. So a regression measured in the
106
+ // first row could vanish from the verdict while the receipt still validated and its
107
+ // hash still verified (an outside correctness audit, finding 1). This is the one
108
+ // definition of an ambiguous receipt: validation, the verdict, the decision and the
109
+ // CLI all call it, so none of them can disagree about what counts.
110
+ function duplicateCaseRows(receipt) {
111
+ const cases = receipt && receipt.results && Array.isArray(receipt.results.cases) ? receipt.results.cases : [];
112
+ const seen = new Map();
113
+ cases.forEach((c, i) => {
114
+ const key = JSON.stringify([c && c.id, c && c.mode]);
115
+ if (!seen.has(key)) seen.set(key, { id: c && c.id, mode: c && c.mode, rows: [] });
116
+ seen.get(key).rows.push(i);
117
+ });
118
+ return [...seen.values()].filter((d) => d.rows.length > 1);
119
+ }
120
+
121
+ // The sentence every refusal of an ambiguous receipt uses, so the CLI, the verdict
122
+ // and the decision say the same thing about the same receipt.
123
+ function ambiguityLine(dups) {
124
+ return dups.map((d) => `case ${JSON.stringify(d.id)} has ${d.rows.length} ${d.mode} rows (results.cases ${d.rows.join(', ')})`).join('; ');
125
+ }
126
+
85
127
  // Validate against the schema matching the receipt's own schema_version (so both
86
128
  // v0.1 and v0.2 receipts validate). Returns { valid, errors, version }.
129
+ //
130
+ // Spec 062: a schema_version that is absent, or is not a version the table holds, is invalid,
131
+ // with no fallback to the current schema. Absent is not read as current: every receipt the
132
+ // runner and the importers write carries one.
87
133
  function validateReceipt(receipt) {
88
- const version = (receipt && receipt.schema_version) || RECEIPT_SCHEMA_VERSION;
134
+ const version = receipt && typeof receipt === 'object' ? receipt.schema_version : undefined;
89
135
  const validate = getValidator(version);
136
+ if (!validate) {
137
+ return { valid: false, errors: [{ instancePath: '/schema_version', message: `unknown schema_version ${JSON.stringify(version === undefined ? null : version)}: not a version this validator holds (${Object.keys(SCHEMA_FILES).join(', ')}); no other schema was tried` }], version };
138
+ }
90
139
  const valid = validate(receipt);
91
- return { valid, errors: valid ? [] : (validate.errors || []), version };
140
+ const errors = valid ? [] : (validate.errors || []);
141
+ // v0.8 (spec 043 AC-2): run.judge.samples and run.counts.judge_samples_per_generation
142
+ // are two spellings of one count; a receipt that gives both must give one value. v0.9
143
+ // keeps both fields, so it keeps the rule.
144
+ if (version === '0.8' || version === '0.9') {
145
+ if (judgeCountsDisagree(receipt)) errors.push({ instancePath: '/run/counts/judge_samples_per_generation', message: 'differs from run.judge.samples' });
146
+ }
147
+ // Spec 050 AC-2: one row per (id, mode), in every schema version. No schema can say
148
+ // it (uniqueness over a pair of fields is outside JSON Schema), so it is said here.
149
+ for (const d of duplicateCaseRows(receipt)) {
150
+ for (const i of d.rows.slice(1)) errors.push({ instancePath: `/results/cases/${i}`, message: `duplicate case id and mode: ${JSON.stringify(d.id)} ${d.mode} is also at /results/cases/${d.rows[0]}` });
151
+ }
152
+ return { valid: errors.length === 0, errors, version };
153
+ }
154
+
155
+ // Spec 062 (register row 4): whether a receipt ran fewer cases than its suite holds. The receipt
156
+ // records the ids it ran (results.cases) and the suite's count (suite.case_count), not the
157
+ // suite's ids, so a receipt narrows its suite when either arm names fewer distinct case ids than
158
+ // the suite counts (R-5). A failed case keeps its rows, so it is not a cut. A
159
+ // narrowed receipt is not a reading of the suite its suite_hash names, and both readers treat it
160
+ // as incomplete (lib/verdict.js receiptVerdict, lib/decision.js decisionState).
161
+ function suiteNarrowed(receipt) {
162
+ const n = receipt && receipt.suite ? receipt.suite.case_count : undefined;
163
+ if (!Number.isInteger(n)) return false;
164
+ const ids = { with_skill: new Set(), baseline: new Set() };
165
+ for (const c of (receipt.results && receipt.results.cases) || []) if (c && ids[c.mode]) ids[c.mode].add(c.id);
166
+ return ids.with_skill.size < n || ids.baseline.size < n;
167
+ }
168
+
169
+ // ── receipt files (spec 062, register row 3) ────────────────────────────────
170
+ // The writer's naming rule and the readers' listing rule, in one module. Until this the name was
171
+ // assembled in bin/driftproof and read back by lib/decision.js with a different rule: a second run
172
+ // on the same day overwrote the first, a regrade sidecar in the directory was read as a receipt
173
+ // that did not verify, and a model id with a dot was looked for in a name that held its slug.
174
+
175
+ // The file-name form of a skill name or a model id. Named for what it is rather than `slug`: the
176
+ // probe-copies rule reads an exported name, and spec 054's probe has a heading-anchor `slug` of its
177
+ // own that is not this function (spec 062 A-062-1).
178
+ function fileSlug(s) { return String(s).toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-|-$/g, ''); }
179
+
180
+ // <skill>-<model>[-<tag>]-<date>[-<hash12>]. `hash` is the first 12 characters of the receipt's
181
+ // own receipt_hash for a run, so two runs on one day name two files; an import passes its source
182
+ // document's hash, which keeps the name spec 049 gave it.
183
+ function receiptBaseName({ skill, model, date, hash = null, tag = null }) {
184
+ return [fileSlug(skill), fileSlug(model), ...(tag ? [tag] : []), date, ...(hash ? [String(hash).slice(0, 12)] : [])].join('-');
185
+ }
186
+
187
+ // The regrade provenance sidecar: its name is the receipt's base name with this suffix, and its
188
+ // `format` names this version. `regrade`'s writers read both from here (lib/regrade.js the format,
189
+ // bin/driftproof the name), so the listing rule below is the writer's own (spec 069).
190
+ const REGRADE_SIDECAR = { suffix: '.regrade.json', format: 'driftproof-regrade/2' };
191
+ const REGRADE_FAMILY = REGRADE_SIDECAR.format.split('/')[0];
192
+
193
+ // The summary export kept beside its receipt: `export --out <dir>` names it the receipt's base name
194
+ // with this suffix, and lib/export.js stamps this format on it (spec 107). It carries the receipt's
195
+ // receipt_hash as a reference, so the listing rule below skips it by name and shape.
196
+ const SUMMARY_SIDECAR = { suffix: '.summary.json', format: 'driftproof/summary' };
197
+
198
+ // The files a writer puts beside a receipt that carry its receipt_hash and are not receipts. A file
199
+ // is one of these only when its name AND its shape match (spec 069, the operator's ruling Q1): a
200
+ // name alone or a shape alone is not enough, so a receipt renamed, or a sidecar renamed, is not
201
+ // passed over. None of them carries `results`.
202
+ // - regrade provenance: `format` is a version of the regrade format. The published Report 011
203
+ // sidecars carry the version before the writer's current one, and read as sidecars.
204
+ // - surface record (spec 042's run): `receipt` names the receipt it sits beside, which is its own
205
+ // name with the suffix replaced by `.json`.
206
+ // - summary export (spec 107): `format` is the summary format and `format_version` a string of
207
+ // digits.
208
+ const RECEIPT_SIDECARS = [
209
+ { kind: 'regrade provenance', suffix: REGRADE_SIDECAR.suffix,
210
+ shape: (doc) => typeof doc.format === 'string' && new RegExp(`^${REGRADE_FAMILY}/[0-9]+$`).test(doc.format) },
211
+ { kind: 'surface record', suffix: '.surface.json',
212
+ shape: (doc, name) => doc.receipt === `${name.slice(0, -'.surface.json'.length)}.json` },
213
+ { kind: 'summary export', suffix: SUMMARY_SIDECAR.suffix,
214
+ shape: (doc) => doc.format === SUMMARY_SIDECAR.format && typeof doc.format_version === 'string' && /^[0-9]+$/.test(doc.format_version) },
215
+ ];
216
+ function sidecarKind(name, doc) {
217
+ if (Object.hasOwn(doc, 'results')) return null;
218
+ const s = RECEIPT_SIDECARS.find((x) => name.endsWith(x.suffix) && name.length > x.suffix.length && x.shape(doc, name));
219
+ return s ? s.kind : null;
220
+ }
221
+
222
+ // Every receipt in a directory, sorted by file name: { file, receipt, error }. A `.json` file that
223
+ // parses to an object carrying `receipt_hash` or `results` is returned, unless it is a sidecar by
224
+ // name and shape (above), so the caller's hash check refuses by name a receipt with either field
225
+ // removed, rather than reading it as absent (spec 062 A-062-3 for `receipt_hash`, spec 069 for
226
+ // `results`). An object carrying neither is not a receipt and is not returned: a badge, a stale
227
+ // document. A `.json` file that does not parse IS returned, with receipt null and the
228
+ // parser's message: it may be a receipt, and spec 030 AC-2 fails its model closed as unreadable.
229
+ function listReceipts(dir) {
230
+ const out = [];
231
+ for (const name of fs.readdirSync(dir).filter((f) => f.endsWith('.json')).sort()) {
232
+ const file = path.join(dir, name);
233
+ let receipt;
234
+ try { receipt = JSON.parse(fs.readFileSync(file, 'utf8')); } catch (e) { out.push({ file, receipt: null, error: e.message }); continue; }
235
+ if (!receipt || typeof receipt !== 'object' || Array.isArray(receipt)) continue;
236
+ if (!Object.hasOwn(receipt, 'results') && !Object.hasOwn(receipt, 'receipt_hash')) continue;
237
+ if (sidecarKind(name, receipt)) continue;
238
+ out.push({ file, receipt, error: null });
239
+ }
240
+ return out;
92
241
  }
93
242
 
94
243
  function mean(nums) {
@@ -224,11 +373,19 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
224
373
  surface: run.surface,
225
374
  runner_version: run.runner_version,
226
375
  date_utc: run.date_utc,
376
+ // v0.8 (spec 043 AC-4): when generation began, stamped by the runner before its
377
+ // first generation call; date_utc keeps its meaning.
378
+ ...(run.generated_at ? { generated_at: run.generated_at } : {}),
379
+ // v0.8 (spec 043 AC-4; spec 044): a re-judge of frozen outputs. Its archived arms
380
+ // are carried before placeCounts reads them, and the two clocks travel together.
381
+ ...(run.arms ? { arms: run.arms } : {}),
382
+ ...(run.judged_at ? { judged_at: run.judged_at, grader_revision: run.grader_revision } : {}),
227
383
  // v0.3: registry provenance + transcript-retention mode. Defaults keep the
228
384
  // honest, cheapest interpretation when a caller omits them.
229
385
  registry: run.registry || 'unregistered',
230
386
  transcripts: run.transcripts || 'hashes-only',
231
- judge: run.judge || { samples: 1, temperature: null, sampling: 'single', surface: run.surface },
387
+ // v0.8: no default judge-sample count; a caller that says nothing leaves it unknown.
388
+ judge: run.judge || { temperature: null, sampling: 'single', surface: run.surface },
232
389
  // v0.6 (spec 026 AC-1, AC-2): what answered. Carried from the caller as
233
390
  // given and never defaulted: a receipt that does not say what answered it
234
391
  // is refused by the schema, which is the point.
@@ -255,6 +412,8 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
255
412
  // skill.tokens — estimated SKILL.md token size (value-per-token axis).
256
413
  if (run.surface_overhead_note) receipt.run.surface_overhead_note = run.surface_overhead_note;
257
414
  if (skill.tokens != null) receipt.skill.tokens = skill.tokens;
415
+ // v0.9, spec 053: the harness the runner read as the run began ({name, version}; absent on a stub run).
416
+ if (run.harness && typeof run.harness.name === 'string') receipt.run.harness = { name: run.harness.name, version: run.harness.version == null ? null : String(run.harness.version) };
258
417
  // v0.4 economics (additive-optional): the frozen prices this receipt's derived
259
418
  // dollar figures were computed from, and the derived block itself. A receipt
260
419
  // from a surface that reports no usage simply omits both.
@@ -266,9 +425,22 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
266
425
  receipt.run.failed_case_count = failedCount;
267
426
  }
268
427
  if (editorialReviews && editorialReviews.length) receipt.editorial_reviews = editorialReviews;
428
+ // v0.8 counts (spec 043 AC-3, AC-7): what the runner establishes, per case. The judge
429
+ // count is the samples the run was asked for; the generation count is the draws the
430
+ // case recorded. A case without a draw set (failed, or a legacy caller) establishes
431
+ // no generation count, so none is written for it.
432
+ const judgeN = receipt.run.judge && Number.isInteger(receipt.run.judge.samples) ? receipt.run.judge.samples : null;
433
+ placeCounts(receipt, cases.map((c, index) => ({
434
+ index,
435
+ counts: {
436
+ ...(c.generation && Array.isArray(c.generation.draws) ? { generations_per_arm: c.generation.draws.length } : {}),
437
+ ...(judgeN !== null && !caseFailed(c) ? { judge_samples_per_generation: judgeN } : {}),
438
+ },
439
+ })));
269
440
  return sealReceipt(receipt);
270
441
  }
271
442
 
272
443
  module.exports = {
273
444
  buildReceipt, sealReceipt, computeReceiptHash, verifyReceiptHash, validateReceipt, BAND_RULE, comparisonOf, caseFailed, FAILED_STATUSES,
445
+ duplicateCaseRows, ambiguityLine, suiteNarrowed, fileSlug, receiptBaseName, listReceipts, REGRADE_SIDECAR, SUMMARY_SIDECAR,
274
446
  };
package/lib/regrade.js ADDED
@@ -0,0 +1,266 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // Regrade: the executor behind lib/reuse.js's `regrade` decision (spec 044).
5
+ //
6
+ // When only the judge (or the grading template) differs between two receipts,
7
+ // triage says the existing generations can be rescored. This does it: it takes a
8
+ // sealed receipt, the skill it was run on and the answers its draws graded, and
9
+ // judges every draw again with the judge it is given. The generation side of the
10
+ // new receipt is the original's, draw for draw; the judge side is new.
11
+ //
12
+ // THE SAME CODE A RUN JUDGES WITH. Each draw goes through lib/run.js judgeCase,
13
+ // its replies through lib/run.js attest, each case through lib/sampling.js
14
+ // acrossDraws, and the receipt through lib/receipt.js buildReceipt. Nothing here
15
+ // grades, aggregates or seals by a second route.
16
+ //
17
+ // THE DRAW SET IS THE ORIGINAL'S. The sampling rule's escalation decides how many
18
+ // generations to draw; with the generations fixed there is nothing for it to
19
+ // decide, so `n_planned` and `stopping_reason` are carried and no draw is added.
20
+ //
21
+ // A DRAW WITH NO ANSWER TO GRADE IS CARRIED, NOT GRADED: one with no
22
+ // generation_hash (the generation timed out) or one cut at the output cap (spec
23
+ // 026 AC-8). A draw whose original judge gave no score has an answer and is graded.
24
+
25
+ const { sha256 } = require('./canonical');
26
+ const { judgeCase, attest, outcomeFor, resolveCallTimeoutMs } = require('./run');
27
+ const { judgeSettings, promptTemplateHash, rubricHash } = require('./judge');
28
+ const { buildReceipt, verifyReceiptHash, FAILED_STATUSES, REGRADE_SIDECAR } = require('./receipt');
29
+ const { acrossDraws } = require('./sampling');
30
+ const { resolveModel, surfaceForModel, isMeteredSurface } = require('./provider');
31
+ const { priceForModel, assertRegistered } = require('./models');
32
+ const { perCallCostUSD } = require('./cost');
33
+ const { buildPricingSnapshot, computeEconomics } = require('./value');
34
+ const { canonicalModelId } = require('./usage');
35
+ const { RUNNER_VERSION } = require('../config');
36
+
37
+ const isTimeout = (e) => !!(e && (e.code === 'TIMEOUT' || /tim(e|ed)\s*out/i.test(String((e && e.message) || ''))));
38
+
39
+ // Whether a draw carries an answer the judge can grade.
40
+ function gradable(d) { return !!(d && d.generation_hash && d.truncated !== true); }
41
+
42
+ // Every draw to grade, with its case and mode, in receipt order.
43
+ function drawsToGrade(receipt) {
44
+ const out = [];
45
+ for (const c of receipt.results.cases) for (const d of ((c.generation || {}).draws || [])) if (gradable(d)) out.push({ c, d });
46
+ return out;
47
+ }
48
+
49
+ // The answer check (AC-4): every draw to grade has an answer, and each answer is
50
+ // the text its generation_hash was taken over. An answer that fails either is a
51
+ // regrade of something no model wrote.
52
+ function answerProblems(receipt, answers) {
53
+ const bad = [];
54
+ for (const { c, d } of drawsToGrade(receipt)) {
55
+ const where = `${c.id}/${c.mode} draw ${d.draw_index}`;
56
+ const text = answers[d.generation_hash];
57
+ if (typeof text !== 'string') bad.push(`${where}: no answer for generation_hash ${d.generation_hash.slice(0, 12)}`);
58
+ else if (sha256(text) !== d.generation_hash) bad.push(`${where}: the answer's sha256 is not its generation_hash ${d.generation_hash.slice(0, 12)}`);
59
+ }
60
+ return bad;
61
+ }
62
+
63
+ // Everything that must hold before the first call, and what the regrade will
64
+ // cost. No call is made. -> { problems, draws, calls, usd }
65
+ function planRegrade({ receipt, skill, answers, judgeModel, samples }) {
66
+ const problems = [];
67
+ if (!verifyReceiptHash(receipt)) problems.push('the receipt_hash does not verify: the receipt was edited after it was sealed');
68
+ if (skill.contentHash !== receipt.skill.content_hash) problems.push(`the skill's content_hash ${String(skill.contentHash).slice(0, 12)} is not the receipt's ${String(receipt.skill.content_hash).slice(0, 12)}`);
69
+ if (skill.suite.suiteHash !== receipt.suite.suite_hash) problems.push(`the suite's hash ${String(skill.suite.suiteHash).slice(0, 12)} is not the receipt's ${String(receipt.suite.suite_hash).slice(0, 12)}`);
70
+ // One row per (case, mode) (spec 044 A-044-4): a receipt with two rows for one case and mode
71
+ // is ambiguous, which spec 050's validation refuses; refused here before any call, not after the
72
+ // spend.
73
+ const rows = new Set();
74
+ for (const c of receipt.results.cases) {
75
+ const key = `${c.id}\u0000${c.mode}`;
76
+ if (rows.has(key)) problems.push(`case ${c.id}/${c.mode} appears in more than one row; the receipt is ambiguous`);
77
+ rows.add(key);
78
+ }
79
+ for (const c of receipt.results.cases) {
80
+ if (!skill.suite.cases.some((k) => k.id === c.id)) problems.push(`case ${c.id} is not in the suite`);
81
+ if (!c.generation || !Array.isArray(c.generation.draws)) problems.push(`case ${c.id}/${c.mode} carries no draw set; a pre-v0.5 receipt cannot be regraded draw by draw`);
82
+ }
83
+ if (!problems.length) problems.push(...answerProblems(receipt, answers));
84
+ const draws = problems.length ? 0 : drawsToGrade(receipt).length;
85
+ const calls = draws * samples;
86
+ const usd = Math.round(calls * perCallCostUSD(resolveModel(judgeModel), 'judge') * 1e4) / 1e4;
87
+ return { problems, draws, calls, usd };
88
+ }
89
+
90
+ // The generation side of a draw, carried as it was.
91
+ function generationSide(d) {
92
+ const g = { stop_reason: d.stop_reason === undefined ? null : d.stop_reason, truncated: d.truncated === true, reported_model: d.reported_model === undefined ? null : d.reported_model };
93
+ return g;
94
+ }
95
+
96
+ // Regrade one sealed receipt. opts: { trusted, budget, timeoutMs, onProgress, nowIso }.
97
+ // -> { receipt, provenance, calls }
98
+ async function regradeReceipt({ receipt, skill, answers, judgeModel, samples, opts = {} }) {
99
+ const judge = resolveModel(judgeModel);
100
+ const modelId = receipt.run.model_id;
101
+ assertRegistered(judge, 'judge model');
102
+ const trusted = !!opts.trusted;
103
+ const budget = opts.budget || null;
104
+ const onProgress = opts.onProgress || (() => {});
105
+ const timeoutMs = resolveCallTimeoutMs(surfaceForModel(judge), opts);
106
+ const judgedAt = new Date().toISOString();
107
+
108
+ let calls = 0;
109
+ const replies = [];
110
+ const cases = [];
111
+ for (const orig of receipt.results.cases) {
112
+ const caseObj = skill.suite.cases.find((k) => k.id === orig.id);
113
+ const mode = orig.mode;
114
+ const draws = [];
115
+ let last = null;
116
+ for (const d of orig.generation.draws) {
117
+ if (!gradable(d)) { draws.push(JSON.parse(JSON.stringify(d))); continue; }
118
+ const gen = generationSide(d);
119
+ const usage = d.usage ? { usage: d.usage } : {};
120
+ onProgress({ case: orig.id, mode, phase: 'judge', draw: d.draw_index, samples });
121
+ let jr;
122
+ try {
123
+ jr = await judgeCase({ caseObj, response: answers[d.generation_hash], generationHash: d.generation_hash, judgeModel: judge, mode, timeoutMs, samples, trusted });
124
+ } catch (e) {
125
+ if (!isTimeout(e)) throw e;
126
+ if (budget) budget.add((e.judgeAttempts || 1) * perCallCostUSD(judge, 'judge'));
127
+ draws.push({ draw_index: d.draw_index, generation_hash: d.generation_hash, status: 'unmeasured', reason: String((e && e.message) || 'timeout').slice(0, 200), samples: [], mean: null, stddev: null, ...gen, ...usage });
128
+ continue;
129
+ }
130
+ calls += samples;
131
+ for (const reply of (jr.replies || [])) {
132
+ if (!reply) continue;
133
+ replies.push(reply);
134
+ attest(reply, judgeModel, 'judge');
135
+ }
136
+ if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judge, 'judge'));
137
+ if (jr.unmeasured) {
138
+ const u = { draw_index: d.draw_index, generation_hash: d.generation_hash, status: 'unmeasured', reason: String(jr.reason || '').slice(0, 200), samples: [], mean: null, stddev: null, ...gen };
139
+ if ((jr.sampleHashes || []).length) u.judge_sample_hashes = jr.sampleHashes;
140
+ Object.assign(u, usage);
141
+ if (jr.judge_usage) u.judge_usage = jr.judge_usage;
142
+ draws.push(u);
143
+ continue;
144
+ }
145
+ const m = {
146
+ draw_index: d.draw_index, generation_hash: d.generation_hash, status: 'measured',
147
+ samples: jr.caseResult.samples, judge_sample_hashes: jr.caseResult.judge_sample_hashes,
148
+ mean: jr.caseResult.mean, stddev: jr.caseResult.stddev, ...gen, ...usage,
149
+ };
150
+ if (jr.caseResult.judge_usage) m.judge_usage = jr.caseResult.judge_usage;
151
+ draws.push(m);
152
+ last = jr;
153
+ }
154
+ const agg = acrossDraws(draws);
155
+ const generation = {
156
+ n_planned: orig.generation.n_planned,
157
+ n_drawn: agg.n_drawn,
158
+ n_measured: agg.n_measured,
159
+ n_unmeasured: agg.n_unmeasured,
160
+ stopping_reason: orig.generation.stopping_reason,
161
+ mean: agg.mean,
162
+ sd: agg.sd,
163
+ judge_sd_mean: agg.judge_sd_mean,
164
+ variance_ratio: agg.variance_ratio,
165
+ variance_ratio_unavailable: agg.variance_ratio_unavailable,
166
+ n_truncated: draws.filter((x) => x.truncated === true).length,
167
+ draws,
168
+ };
169
+ if (!last) {
170
+ const nonTimeout = draws.some((x) => x.status === 'unmeasured' && !/tim(e|ed)\s*out/i.test(String(x.reason || '')));
171
+ const reason = [...draws].reverse().map((x) => x.reason).find(Boolean) || 'timeout';
172
+ cases.push({ id: orig.id, mode, case_status: nonTimeout ? FAILED_STATUSES[1] : FAILED_STATUSES[0], reason, generation });
173
+ continue;
174
+ }
175
+ const caseResult = { ...last.caseResult, generation };
176
+ caseResult.mean = agg.mean;
177
+ caseResult.score = agg.mean;
178
+ caseResult.stddev = agg.sd;
179
+ caseResult.outcome = outcomeFor(agg.mean, agg.sd, caseObj.pass_threshold);
180
+ onProgress({ case: orig.id, mode, phase: 'done', outcome: caseResult.outcome, score: caseResult.mean, stddev: caseResult.stddev });
181
+ cases.push(caseResult);
182
+ }
183
+
184
+ // What answered the judge. A stub judge graded nothing: its receipt is UNVERIFIED
185
+ // with judge surface stub, whatever answered the generations (spec 026 AC-1).
186
+ const judgeStub = replies.length > 0 && replies.every((r) => r.answeredBy === 'stub');
187
+ const judgeModelAnswered = replies.length > 0 && replies.every((r) => r.answeredBy === 'model');
188
+ const judgeBlock = { ...judgeSettings(samples, judge), ...(judgeStub ? { surface: 'stub' } : {}), model_id: judge, prompt_template_hash: promptTemplateHash() };
189
+ const nowIso = opts.nowIso || new Date().toISOString();
190
+ const surface = receipt.run.surface;
191
+ const pricingSnapshot = buildPricingSnapshot({ models: [modelId, judge], lookup: priceForModel, nowIso });
192
+ const economics = computeEconomics({ cases, modelId, judgeModelId: judge, pricingSnapshot, surface, meteredSurface: isMeteredSurface(surface) });
193
+
194
+ // v0.8 (spec 043 AC-4). Every arm was generated in an earlier run, so each is archived:
195
+ // its own model and generation time, carried from the original (the original's own arm
196
+ // override where it has one, else its run's), and its own counts, because an archived arm
197
+ // never inherits run.counts. lib/counts.js placeCounts places counts for arms generated in
198
+ // this run and none is, so they are placed here by its rule: at arm scope when every case
199
+ // of the arm has the same count, else on each case that has one. Nothing is generated in
200
+ // a regrade, so the receipt carries no run.generated_at and no run.counts.
201
+ const arms = {};
202
+ for (const mode of ['with_skill', 'baseline']) {
203
+ const modeCases = cases.filter((c) => c.mode === mode);
204
+ if (!modeCases.length) continue;
205
+ const a = (receipt.run.arms && receipt.run.arms[mode]) || {};
206
+ arms[mode] = { model_id: a.model_id || modelId, generated_at: a.generated_at || receipt.run.generated_at || receipt.run.date_utc };
207
+ const perCase = modeCases.map((c) => ({
208
+ c,
209
+ counts: {
210
+ ...(c.generation && Array.isArray(c.generation.draws) ? { generations_per_arm: c.generation.draws.length } : {}),
211
+ ...(!c.case_status ? { judge_samples_per_generation: samples } : {}),
212
+ },
213
+ }));
214
+ for (const kind of ['generations_per_arm', 'judge_samples_per_generation']) {
215
+ const values = perCase.map((x) => (Number.isInteger(x.counts[kind]) ? x.counts[kind] : null));
216
+ if (values.every((v) => v !== null) && new Set(values).size === 1) arms[mode].counts = { ...(arms[mode].counts || {}), [kind]: values[0] };
217
+ else for (const x of perCase) if (Number.isInteger(x.counts[kind])) x.c.counts = { ...(x.c.counts || {}), [kind]: x.counts[kind] };
218
+ }
219
+ }
220
+ const graderRevision = {
221
+ prompt_template_hash: judgeBlock.prompt_template_hash,
222
+ rubric_hashes: cases.map((c) => { const k = skill.suite.cases.find((x) => x.id === c.id); return k ? rubricHash(k.rubric) : null; }),
223
+ };
224
+ const out = buildReceipt({
225
+ skill: { name: receipt.skill.name, version: receipt.skill.version, contentHash: receipt.skill.content_hash, tokens: receipt.skill.tokens },
226
+ suite: { format: receipt.suite.format, suiteHash: receipt.suite.suite_hash, caseCount: receipt.suite.case_count, canary: receipt.suite.canary },
227
+ run: {
228
+ model_id: modelId,
229
+ model_release_date: receipt.run.model_release_date,
230
+ provider: receipt.run.provider,
231
+ surface,
232
+ surface_overhead_note: receipt.run.surface_overhead_note,
233
+ runner_version: RUNNER_VERSION,
234
+ date_utc: nowIso,
235
+ registry: receipt.run.registry,
236
+ transcripts: 'hashes-only',
237
+ judge: judgeBlock,
238
+ pricing_snapshot: pricingSnapshot,
239
+ answered_by: receipt.run.answered_by,
240
+ arms,
241
+ judged_at: judgedAt,
242
+ grader_revision: graderRevision,
243
+ },
244
+ cases,
245
+ economics,
246
+ verificationLevel: receipt.verification_level === 'TESTED' && judgeModelAnswered ? 'TESTED' : 'UNVERIFIED',
247
+ });
248
+
249
+ // Where the regrade came from: what v0.8 has no field for. The judge time, the grader
250
+ // revision and the archived arms are in the receipt (spec 043 AC-4) and are not repeated
251
+ // here; this sidecar is bound to the receipt by its hash.
252
+ const provenance = {
253
+ format: REGRADE_SIDECAR.format,
254
+ receipt_hash: out.receipt_hash,
255
+ regraded_from: { receipt_hash: receipt.receipt_hash, runner_version: receipt.run.runner_version, date_utc: receipt.run.date_utc, judge: receipt.run.judge },
256
+ generated_at_basis: receipt.run.generated_at || (receipt.run.arms && Object.values(receipt.run.arms).some((x) => x && x.generated_at))
257
+ ? "the original receipt's own generated_at"
258
+ : "the original receipt's run.date_utc: a receipt of that schema records no separate generation time",
259
+ draws: { graded: drawsToGrade(receipt).length, carried: receipt.results.cases.reduce((a, c) => a + c.generation.draws.length, 0) - drawsToGrade(receipt).length },
260
+ judge_reported_models: [...new Set(replies.flatMap((r) => (r.reportedModels || []).map((id) => canonicalModelId(id))))].sort(),
261
+ calls,
262
+ };
263
+ return { receipt: out, provenance, calls };
264
+ }
265
+
266
+ module.exports = { regradeReceipt, planRegrade, answerProblems, drawsToGrade, gradable };