driftproof 0.11.1 → 0.11.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/importers.js CHANGED
@@ -22,6 +22,7 @@ const { mean, stddev, round, combineUncertainty, aggregateBands } = require('./s
22
22
  const { outcomeFor } = require('./run');
23
23
  const { inferProvider } = require('./provider');
24
24
  const { registryStatus } = require('./models');
25
+ const { placeCounts } = require('./counts');
25
26
 
26
27
  const IMPORT_TOOLS = ['agent-skills-eval', 'skillgrade'];
27
28
 
@@ -41,10 +42,17 @@ function aggregateMode(cases) {
41
42
  }
42
43
 
43
44
  // Shared receipt shell for both importers. `judgeBlock` describes the SOURCE
44
- // tool's grading (samples = what it actually did), never our sampled judge.
45
- function importedReceipt({ tool, skillName, skillVersion, suiteFormat, caseCount, modelId, dateUtc, judgeBlock, cases, comparison }) {
46
- const withSkill = cases.filter((c) => c.mode === 'with_skill');
47
- const baseline = cases.filter((c) => c.mode === 'baseline');
45
+ // tool's grading, never our sampled judge. v0.8 (spec 043): it carries no sample
46
+ // count, because neither source format defines one; `perCase` carries only the counts
47
+ // the source establishes, and placeCounts writes them at the narrowest honest scope.
48
+ // A case with no observation (case_status no_observations) is listed, excluded from
49
+ // every aggregate, and named in excluded_cases with `excludedReasons[id]`.
50
+ function importedReceipt({ tool, skillName, skillVersion, suiteFormat, caseCount, modelId, dateUtc, judgeBlock, cases, comparison, perCase = [], excludedReasons = {} }) {
51
+ const measured = cases.filter((c) => !c.case_status || c.case_status === 'ok');
52
+ const withSkill = measured.filter((c) => c.mode === 'with_skill');
53
+ const baseline = measured.filter((c) => c.mode === 'baseline');
54
+ const excluded = cases.filter((c) => c.case_status && c.case_status !== 'ok')
55
+ .map((c) => ({ id: c.id, modes: [c.mode], reason: excludedReasons[c.id] || 'the source supplied no observation for this case' }));
48
56
  const receipt = {
49
57
  schema_version: RECEIPT_SCHEMA_VERSION,
50
58
  skill: {
@@ -73,12 +81,13 @@ function importedReceipt({ tool, skillName, skillVersion, suiteFormat, caseCount
73
81
  },
74
82
  results: {
75
83
  cases,
76
- aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline), band_rule: BAND_RULE },
84
+ aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline), band_rule: BAND_RULE, ...(excluded.length ? { excluded_cases: excluded } : {}) },
77
85
  },
78
86
  comparison,
79
87
  verification_level: 'DECLARED',
80
88
  receipt_hash: '',
81
89
  };
90
+ placeCounts(receipt, perCase);
82
91
  return sealReceipt(receipt);
83
92
  }
84
93
 
@@ -139,7 +148,9 @@ function importAgentSkillsEval(data, { importedAt } = {}) {
139
148
  caseCount: data.evals.length,
140
149
  modelId: data.target || 'unknown',
141
150
  dateUtc: data.timestamp || importedAt || new Date().toISOString(),
142
- judgeBlock: { samples: 1, temperature: null, sampling: 'external', surface: 'external' },
151
+ // No count of either kind: the agent-skills-eval benchmark format defines neither
152
+ // (spec 043 AC-5, A-1), so none is written until it does.
153
+ judgeBlock: { temperature: null, sampling: 'external', surface: 'external' },
143
154
  cases,
144
155
  comparison,
145
156
  });
@@ -158,16 +169,38 @@ function importSkillgrade(data, { importedAt } = {}) {
158
169
  const graderModel = data.grader_model || 'unknown';
159
170
  const defaultThreshold = typeof data.threshold === 'number' ? data.threshold : 0.8;
160
171
  const cases = [];
161
- let maxTrials = 1;
172
+ const perCase = [];
173
+ const excludedReasons = {};
162
174
  for (const task of data.tasks) {
163
- const rewards = (task.trials || []).map((t) => (typeof t === 'number' ? t : t.reward)).filter((r) => typeof r === 'number');
164
- if (!rewards.length) continue;
165
- maxTrials = Math.max(maxTrials, rewards.length);
175
+ const id = String(task.name);
176
+ // Spec 043 A-3: keep the distinction the source makes. A trials list that is
177
+ // present and empty is zero observations; no trials field is an unknown count.
178
+ // Either way the task is listed and contributes no figure.
179
+ if (!Array.isArray(task.trials)) {
180
+ cases.push({ id, mode: 'with_skill', case_status: 'no_observations' });
181
+ excludedReasons[id] = 'the source carries no trials field for this task: its trial count is unknown';
182
+ continue;
183
+ }
184
+ const rewards = task.trials.map((t) => (typeof t === 'number' ? t : t.reward)).filter((r) => typeof r === 'number');
185
+ if (!task.trials.length) {
186
+ perCase.push({ index: cases.length, counts: { generations_per_arm: 0 } });
187
+ cases.push({ id, mode: 'with_skill', case_status: 'no_observations', samples: [] });
188
+ excludedReasons[id] = 'the source lists no trials for this task (trials is empty): zero observations';
189
+ continue;
190
+ }
191
+ if (!rewards.length) {
192
+ cases.push({ id, mode: 'with_skill', case_status: 'no_observations' });
193
+ excludedReasons[id] = 'the source lists trials for this task but none carries a numeric reward';
194
+ continue;
195
+ }
196
+ // Trials are independent generations (A-2): their number is this case's
197
+ // generation count, and never a judge-sample count.
198
+ perCase.push({ index: cases.length, counts: { generations_per_arm: rewards.length } });
166
199
  const m = round(mean(rewards));
167
200
  const sd = round(stddev(rewards));
168
201
  const threshold = typeof task.threshold === 'number' ? task.threshold : defaultThreshold;
169
202
  cases.push({
170
- id: String(task.name),
203
+ id,
171
204
  mode: 'with_skill',
172
205
  // Same outcome rule as a Driftproof run (borderline when the threshold
173
206
  // sits inside mean ± stddev) — a deterministic READING of their numbers.
@@ -188,11 +221,15 @@ function importSkillgrade(data, { importedAt } = {}) {
188
221
  // imported verbatim (or the results' model field when present).
189
222
  modelId: data.model || data.agent || 'unknown',
190
223
  dateUtc: data.timestamp || importedAt || new Date().toISOString(),
191
- judgeBlock: { samples: maxTrials, temperature: null, sampling: 'external', surface: 'external' },
224
+ // No judge-sample count: the skillgrade format defines none, and a trial count is a
225
+ // generation count (spec 043 AC-5, A-1, A-2).
226
+ judgeBlock: { temperature: null, sampling: 'external', surface: 'external' },
192
227
  cases,
228
+ perCase,
229
+ excludedReasons,
193
230
  // No baseline mode exists in skillgrade — nulls, never a fabricated 0.
194
231
  comparison: {
195
- with_skill_score: round(mean(cases.map((c) => c.mean))),
232
+ with_skill_score: round(mean(cases.filter((c) => !c.case_status).map((c) => c.mean))),
196
233
  baseline_score: null,
197
234
  delta: null,
198
235
  delta_uncertainty: null,
package/lib/receipt.js CHANGED
@@ -5,6 +5,7 @@ const fs = require('fs');
5
5
  const path = require('path');
6
6
  const { canonicalize, sha256 } = require('./canonical');
7
7
  const { RECEIPT_SCHEMA_VERSION } = require('../config');
8
+ const { placeCounts, judgeCountsDisagree } = require('./counts');
8
9
  const { aggregateBands, combineUncertainty, round } = require('./stats');
9
10
 
10
11
  // Schema file per receipt version. The current schema is receipt.schema.json;
@@ -29,7 +30,15 @@ const SCHEMA_FILES = {
29
30
  // verdict token a reader derives (UNDERPOWERED), and a v0.6 receipt restamped 0.7
30
31
  // is refused by the const, so the number keeps resolving to the schema it meant.
31
32
  '0.6': 'receipt.v0.6.schema.json',
32
- '0.7': 'receipt.schema.json',
33
+ // v0.7 moved the same way when v0.8 took the current pointer (spec 043): v0.8 makes
34
+ // run.judge.samples optional and adds counts and clocks, and its const refuses a v0.7
35
+ // receipt that has not been restamped, so the number keeps resolving to the schema it meant.
36
+ '0.7': 'receipt.v0.7.schema.json',
37
+ // v0.8 moved the same way when v0.9 took the current pointer (spec 049): v0.9 adds the
38
+ // import record, the harness, the unit and activation, and its const refuses a v0.8 receipt
39
+ // that has not been restamped, so the number keeps resolving to the schema it meant.
40
+ '0.8': 'receipt.v0.8.schema.json',
41
+ '0.9': 'receipt.schema.json',
33
42
  };
34
43
 
35
44
  // THE BAND RULE, stated on every receipt (results.aggregates.band_rule, spec
@@ -82,13 +91,50 @@ function verifyReceiptHash(receipt) {
82
91
  return receipt.receipt_hash === computeReceiptHash(receipt);
83
92
  }
84
93
 
94
+ // Spec 050: every (id, mode) pair that occurs more than once in results.cases, with
95
+ // the index of each row that carries it. A receipt is one row per case and arm; two
96
+ // rows for one pair are two measurements that every reader keyed by (id, mode) would
97
+ // collapse into one, keeping whichever came last. So a regression measured in the
98
+ // first row could vanish from the verdict while the receipt still validated and its
99
+ // hash still verified (an outside correctness audit, finding 1). This is the one
100
+ // definition of an ambiguous receipt: validation, the verdict, the decision and the
101
+ // CLI all call it, so none of them can disagree about what counts.
102
+ function duplicateCaseRows(receipt) {
103
+ const cases = receipt && receipt.results && Array.isArray(receipt.results.cases) ? receipt.results.cases : [];
104
+ const seen = new Map();
105
+ cases.forEach((c, i) => {
106
+ const key = JSON.stringify([c && c.id, c && c.mode]);
107
+ if (!seen.has(key)) seen.set(key, { id: c && c.id, mode: c && c.mode, rows: [] });
108
+ seen.get(key).rows.push(i);
109
+ });
110
+ return [...seen.values()].filter((d) => d.rows.length > 1);
111
+ }
112
+
113
+ // The sentence every refusal of an ambiguous receipt uses, so the CLI, the verdict
114
+ // and the decision say the same thing about the same receipt.
115
+ function ambiguityLine(dups) {
116
+ return dups.map((d) => `case ${JSON.stringify(d.id)} has ${d.rows.length} ${d.mode} rows (results.cases ${d.rows.join(', ')})`).join('; ');
117
+ }
118
+
85
119
  // Validate against the schema matching the receipt's own schema_version (so both
86
120
  // v0.1 and v0.2 receipts validate). Returns { valid, errors, version }.
87
121
  function validateReceipt(receipt) {
88
122
  const version = (receipt && receipt.schema_version) || RECEIPT_SCHEMA_VERSION;
89
123
  const validate = getValidator(version);
90
124
  const valid = validate(receipt);
91
- return { valid, errors: valid ? [] : (validate.errors || []), version };
125
+ const errors = valid ? [] : (validate.errors || []);
126
+ // v0.8 (spec 043 AC-2): run.judge.samples and run.counts.judge_samples_per_generation
127
+ // are two spellings of one count; a receipt that gives both must give one value. v0.9
128
+ // keeps both fields, so it keeps the rule.
129
+ if (version === '0.8' || version === '0.9') {
130
+ if (judgeCountsDisagree(receipt)) errors.push({ instancePath: '/run/counts/judge_samples_per_generation', message: 'differs from run.judge.samples' });
131
+ }
132
+ // Spec 050 AC-2: one row per (id, mode), in every schema version. No schema can say
133
+ // it (uniqueness over a pair of fields is outside JSON Schema), so it is said here.
134
+ for (const d of duplicateCaseRows(receipt)) {
135
+ for (const i of d.rows.slice(1)) errors.push({ instancePath: `/results/cases/${i}`, message: `duplicate case id and mode: ${JSON.stringify(d.id)} ${d.mode} is also at /results/cases/${d.rows[0]}` });
136
+ }
137
+ return { valid: errors.length === 0, errors, version };
92
138
  }
93
139
 
94
140
  function mean(nums) {
@@ -224,11 +270,19 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
224
270
  surface: run.surface,
225
271
  runner_version: run.runner_version,
226
272
  date_utc: run.date_utc,
273
+ // v0.8 (spec 043 AC-4): when generation began, stamped by the runner before its
274
+ // first generation call; date_utc keeps its meaning.
275
+ ...(run.generated_at ? { generated_at: run.generated_at } : {}),
276
+ // v0.8 (spec 043 AC-4; spec 044): a re-judge of frozen outputs. Its archived arms
277
+ // are carried before placeCounts reads them, and the two clocks travel together.
278
+ ...(run.arms ? { arms: run.arms } : {}),
279
+ ...(run.judged_at ? { judged_at: run.judged_at, grader_revision: run.grader_revision } : {}),
227
280
  // v0.3: registry provenance + transcript-retention mode. Defaults keep the
228
281
  // honest, cheapest interpretation when a caller omits them.
229
282
  registry: run.registry || 'unregistered',
230
283
  transcripts: run.transcripts || 'hashes-only',
231
- judge: run.judge || { samples: 1, temperature: null, sampling: 'single', surface: run.surface },
284
+ // v0.8: no default judge-sample count; a caller that says nothing leaves it unknown.
285
+ judge: run.judge || { temperature: null, sampling: 'single', surface: run.surface },
232
286
  // v0.6 (spec 026 AC-1, AC-2): what answered. Carried from the caller as
233
287
  // given and never defaulted: a receipt that does not say what answered it
234
288
  // is refused by the schema, which is the point.
@@ -255,6 +309,8 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
255
309
  // skill.tokens — estimated SKILL.md token size (value-per-token axis).
256
310
  if (run.surface_overhead_note) receipt.run.surface_overhead_note = run.surface_overhead_note;
257
311
  if (skill.tokens != null) receipt.skill.tokens = skill.tokens;
312
+ // v0.9, spec 053: the harness the runner read as the run began ({name, version}; absent on a stub run).
313
+ if (run.harness && typeof run.harness.name === 'string') receipt.run.harness = { name: run.harness.name, version: run.harness.version == null ? null : String(run.harness.version) };
258
314
  // v0.4 economics (additive-optional): the frozen prices this receipt's derived
259
315
  // dollar figures were computed from, and the derived block itself. A receipt
260
316
  // from a surface that reports no usage simply omits both.
@@ -266,9 +322,22 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
266
322
  receipt.run.failed_case_count = failedCount;
267
323
  }
268
324
  if (editorialReviews && editorialReviews.length) receipt.editorial_reviews = editorialReviews;
325
+ // v0.8 counts (spec 043 AC-3, AC-7): what the runner establishes, per case. The judge
326
+ // count is the samples the run was asked for; the generation count is the draws the
327
+ // case recorded. A case without a draw set (failed, or a legacy caller) establishes
328
+ // no generation count, so none is written for it.
329
+ const judgeN = receipt.run.judge && Number.isInteger(receipt.run.judge.samples) ? receipt.run.judge.samples : null;
330
+ placeCounts(receipt, cases.map((c, index) => ({
331
+ index,
332
+ counts: {
333
+ ...(c.generation && Array.isArray(c.generation.draws) ? { generations_per_arm: c.generation.draws.length } : {}),
334
+ ...(judgeN !== null && !caseFailed(c) ? { judge_samples_per_generation: judgeN } : {}),
335
+ },
336
+ })));
269
337
  return sealReceipt(receipt);
270
338
  }
271
339
 
272
340
  module.exports = {
273
341
  buildReceipt, sealReceipt, computeReceiptHash, verifyReceiptHash, validateReceipt, BAND_RULE, comparisonOf, caseFailed, FAILED_STATUSES,
342
+ duplicateCaseRows, ambiguityLine,
274
343
  };
package/lib/regrade.js ADDED
@@ -0,0 +1,266 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // Regrade: the executor behind lib/reuse.js's `regrade` decision (spec 044).
5
+ //
6
+ // When only the judge (or the grading template) differs between two receipts,
7
+ // triage says the existing generations can be rescored. This does it: it takes a
8
+ // sealed receipt, the skill it was run on and the answers its draws graded, and
9
+ // judges every draw again with the judge it is given. The generation side of the
10
+ // new receipt is the original's, draw for draw; the judge side is new.
11
+ //
12
+ // THE SAME CODE A RUN JUDGES WITH. Each draw goes through lib/run.js judgeCase,
13
+ // its replies through lib/run.js attest, each case through lib/sampling.js
14
+ // acrossDraws, and the receipt through lib/receipt.js buildReceipt. Nothing here
15
+ // grades, aggregates or seals by a second route.
16
+ //
17
+ // THE DRAW SET IS THE ORIGINAL'S. The sampling rule's escalation decides how many
18
+ // generations to draw; with the generations fixed there is nothing for it to
19
+ // decide, so `n_planned` and `stopping_reason` are carried and no draw is added.
20
+ //
21
+ // A DRAW WITH NO ANSWER TO GRADE IS CARRIED, NOT GRADED: one with no
22
+ // generation_hash (the generation timed out) or one cut at the output cap (spec
23
+ // 026 AC-8). A draw whose original judge gave no score has an answer and is graded.
24
+
25
+ const { sha256 } = require('./canonical');
26
+ const { judgeCase, attest, outcomeFor, resolveCallTimeoutMs } = require('./run');
27
+ const { judgeSettings, promptTemplateHash, rubricHash } = require('./judge');
28
+ const { buildReceipt, verifyReceiptHash, FAILED_STATUSES } = require('./receipt');
29
+ const { acrossDraws } = require('./sampling');
30
+ const { resolveModel, surfaceForModel, isMeteredSurface } = require('./provider');
31
+ const { priceForModel, assertRegistered } = require('./models');
32
+ const { perCallCostUSD } = require('./cost');
33
+ const { buildPricingSnapshot, computeEconomics } = require('./value');
34
+ const { canonicalModelId } = require('./usage');
35
+ const { RUNNER_VERSION } = require('../config');
36
+
37
+ const isTimeout = (e) => !!(e && (e.code === 'TIMEOUT' || /tim(e|ed)\s*out/i.test(String((e && e.message) || ''))));
38
+
39
+ // Whether a draw carries an answer the judge can grade.
40
+ function gradable(d) { return !!(d && d.generation_hash && d.truncated !== true); }
41
+
42
+ // Every draw to grade, with its case and mode, in receipt order.
43
+ function drawsToGrade(receipt) {
44
+ const out = [];
45
+ for (const c of receipt.results.cases) for (const d of ((c.generation || {}).draws || [])) if (gradable(d)) out.push({ c, d });
46
+ return out;
47
+ }
48
+
49
+ // The answer check (AC-4): every draw to grade has an answer, and each answer is
50
+ // the text its generation_hash was taken over. An answer that fails either is a
51
+ // regrade of something no model wrote.
52
+ function answerProblems(receipt, answers) {
53
+ const bad = [];
54
+ for (const { c, d } of drawsToGrade(receipt)) {
55
+ const where = `${c.id}/${c.mode} draw ${d.draw_index}`;
56
+ const text = answers[d.generation_hash];
57
+ if (typeof text !== 'string') bad.push(`${where}: no answer for generation_hash ${d.generation_hash.slice(0, 12)}`);
58
+ else if (sha256(text) !== d.generation_hash) bad.push(`${where}: the answer's sha256 is not its generation_hash ${d.generation_hash.slice(0, 12)}`);
59
+ }
60
+ return bad;
61
+ }
62
+
63
+ // Everything that must hold before the first call, and what the regrade will
64
+ // cost. No call is made. -> { problems, draws, calls, usd }
65
+ function planRegrade({ receipt, skill, answers, judgeModel, samples }) {
66
+ const problems = [];
67
+ if (!verifyReceiptHash(receipt)) problems.push('the receipt_hash does not verify: the receipt was edited after it was sealed');
68
+ if (skill.contentHash !== receipt.skill.content_hash) problems.push(`the skill's content_hash ${String(skill.contentHash).slice(0, 12)} is not the receipt's ${String(receipt.skill.content_hash).slice(0, 12)}`);
69
+ if (skill.suite.suiteHash !== receipt.suite.suite_hash) problems.push(`the suite's hash ${String(skill.suite.suiteHash).slice(0, 12)} is not the receipt's ${String(receipt.suite.suite_hash).slice(0, 12)}`);
70
+ // One row per (case, mode) (spec 044 A-044-4): a receipt with two rows for one case and mode
71
+ // is ambiguous, which spec 050's validation refuses; refused here before any call, not after the
72
+ // spend.
73
+ const rows = new Set();
74
+ for (const c of receipt.results.cases) {
75
+ const key = `${c.id}\u0000${c.mode}`;
76
+ if (rows.has(key)) problems.push(`case ${c.id}/${c.mode} appears in more than one row; the receipt is ambiguous`);
77
+ rows.add(key);
78
+ }
79
+ for (const c of receipt.results.cases) {
80
+ if (!skill.suite.cases.some((k) => k.id === c.id)) problems.push(`case ${c.id} is not in the suite`);
81
+ if (!c.generation || !Array.isArray(c.generation.draws)) problems.push(`case ${c.id}/${c.mode} carries no draw set; a pre-v0.5 receipt cannot be regraded draw by draw`);
82
+ }
83
+ if (!problems.length) problems.push(...answerProblems(receipt, answers));
84
+ const draws = problems.length ? 0 : drawsToGrade(receipt).length;
85
+ const calls = draws * samples;
86
+ const usd = Math.round(calls * perCallCostUSD(resolveModel(judgeModel), 'judge') * 1e4) / 1e4;
87
+ return { problems, draws, calls, usd };
88
+ }
89
+
90
+ // The generation side of a draw, carried as it was.
91
+ function generationSide(d) {
92
+ const g = { stop_reason: d.stop_reason === undefined ? null : d.stop_reason, truncated: d.truncated === true, reported_model: d.reported_model === undefined ? null : d.reported_model };
93
+ return g;
94
+ }
95
+
96
+ // Regrade one sealed receipt. opts: { trusted, budget, timeoutMs, onProgress, nowIso }.
97
+ // -> { receipt, provenance, calls }
98
+ async function regradeReceipt({ receipt, skill, answers, judgeModel, samples, opts = {} }) {
99
+ const judge = resolveModel(judgeModel);
100
+ const modelId = receipt.run.model_id;
101
+ assertRegistered(judge, 'judge model');
102
+ const trusted = !!opts.trusted;
103
+ const budget = opts.budget || null;
104
+ const onProgress = opts.onProgress || (() => {});
105
+ const timeoutMs = resolveCallTimeoutMs(surfaceForModel(judge), opts);
106
+ const judgedAt = new Date().toISOString();
107
+
108
+ let calls = 0;
109
+ const replies = [];
110
+ const cases = [];
111
+ for (const orig of receipt.results.cases) {
112
+ const caseObj = skill.suite.cases.find((k) => k.id === orig.id);
113
+ const mode = orig.mode;
114
+ const draws = [];
115
+ let last = null;
116
+ for (const d of orig.generation.draws) {
117
+ if (!gradable(d)) { draws.push(JSON.parse(JSON.stringify(d))); continue; }
118
+ const gen = generationSide(d);
119
+ const usage = d.usage ? { usage: d.usage } : {};
120
+ onProgress({ case: orig.id, mode, phase: 'judge', draw: d.draw_index, samples });
121
+ let jr;
122
+ try {
123
+ jr = await judgeCase({ caseObj, response: answers[d.generation_hash], generationHash: d.generation_hash, judgeModel: judge, mode, timeoutMs, samples, trusted });
124
+ } catch (e) {
125
+ if (!isTimeout(e)) throw e;
126
+ if (budget) budget.add((e.judgeAttempts || 1) * perCallCostUSD(judge, 'judge'));
127
+ draws.push({ draw_index: d.draw_index, generation_hash: d.generation_hash, status: 'unmeasured', reason: String((e && e.message) || 'timeout').slice(0, 200), samples: [], mean: null, stddev: null, ...gen, ...usage });
128
+ continue;
129
+ }
130
+ calls += samples;
131
+ for (const reply of (jr.replies || [])) {
132
+ if (!reply) continue;
133
+ replies.push(reply);
134
+ attest(reply, judgeModel, 'judge');
135
+ }
136
+ if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judge, 'judge'));
137
+ if (jr.unmeasured) {
138
+ const u = { draw_index: d.draw_index, generation_hash: d.generation_hash, status: 'unmeasured', reason: String(jr.reason || '').slice(0, 200), samples: [], mean: null, stddev: null, ...gen };
139
+ if ((jr.sampleHashes || []).length) u.judge_sample_hashes = jr.sampleHashes;
140
+ Object.assign(u, usage);
141
+ if (jr.judge_usage) u.judge_usage = jr.judge_usage;
142
+ draws.push(u);
143
+ continue;
144
+ }
145
+ const m = {
146
+ draw_index: d.draw_index, generation_hash: d.generation_hash, status: 'measured',
147
+ samples: jr.caseResult.samples, judge_sample_hashes: jr.caseResult.judge_sample_hashes,
148
+ mean: jr.caseResult.mean, stddev: jr.caseResult.stddev, ...gen, ...usage,
149
+ };
150
+ if (jr.caseResult.judge_usage) m.judge_usage = jr.caseResult.judge_usage;
151
+ draws.push(m);
152
+ last = jr;
153
+ }
154
+ const agg = acrossDraws(draws);
155
+ const generation = {
156
+ n_planned: orig.generation.n_planned,
157
+ n_drawn: agg.n_drawn,
158
+ n_measured: agg.n_measured,
159
+ n_unmeasured: agg.n_unmeasured,
160
+ stopping_reason: orig.generation.stopping_reason,
161
+ mean: agg.mean,
162
+ sd: agg.sd,
163
+ judge_sd_mean: agg.judge_sd_mean,
164
+ variance_ratio: agg.variance_ratio,
165
+ variance_ratio_unavailable: agg.variance_ratio_unavailable,
166
+ n_truncated: draws.filter((x) => x.truncated === true).length,
167
+ draws,
168
+ };
169
+ if (!last) {
170
+ const nonTimeout = draws.some((x) => x.status === 'unmeasured' && !/tim(e|ed)\s*out/i.test(String(x.reason || '')));
171
+ const reason = [...draws].reverse().map((x) => x.reason).find(Boolean) || 'timeout';
172
+ cases.push({ id: orig.id, mode, case_status: nonTimeout ? FAILED_STATUSES[1] : FAILED_STATUSES[0], reason, generation });
173
+ continue;
174
+ }
175
+ const caseResult = { ...last.caseResult, generation };
176
+ caseResult.mean = agg.mean;
177
+ caseResult.score = agg.mean;
178
+ caseResult.stddev = agg.sd;
179
+ caseResult.outcome = outcomeFor(agg.mean, agg.sd, caseObj.pass_threshold);
180
+ onProgress({ case: orig.id, mode, phase: 'done', outcome: caseResult.outcome, score: caseResult.mean, stddev: caseResult.stddev });
181
+ cases.push(caseResult);
182
+ }
183
+
184
+ // What answered the judge. A stub judge graded nothing: its receipt is UNVERIFIED
185
+ // with judge surface stub, whatever answered the generations (spec 026 AC-1).
186
+ const judgeStub = replies.length > 0 && replies.every((r) => r.answeredBy === 'stub');
187
+ const judgeModelAnswered = replies.length > 0 && replies.every((r) => r.answeredBy === 'model');
188
+ const judgeBlock = { ...judgeSettings(samples, judge), ...(judgeStub ? { surface: 'stub' } : {}), model_id: judge, prompt_template_hash: promptTemplateHash() };
189
+ const nowIso = opts.nowIso || new Date().toISOString();
190
+ const surface = receipt.run.surface;
191
+ const pricingSnapshot = buildPricingSnapshot({ models: [modelId, judge], lookup: priceForModel, nowIso });
192
+ const economics = computeEconomics({ cases, modelId, judgeModelId: judge, pricingSnapshot, surface, meteredSurface: isMeteredSurface(surface) });
193
+
194
+ // v0.8 (spec 043 AC-4). Every arm was generated in an earlier run, so each is archived:
195
+ // its own model and generation time, carried from the original (the original's own arm
196
+ // override where it has one, else its run's), and its own counts, because an archived arm
197
+ // never inherits run.counts. lib/counts.js placeCounts places counts for arms generated in
198
+ // this run and none is, so they are placed here by its rule: at arm scope when every case
199
+ // of the arm has the same count, else on each case that has one. Nothing is generated in
200
+ // a regrade, so the receipt carries no run.generated_at and no run.counts.
201
+ const arms = {};
202
+ for (const mode of ['with_skill', 'baseline']) {
203
+ const modeCases = cases.filter((c) => c.mode === mode);
204
+ if (!modeCases.length) continue;
205
+ const a = (receipt.run.arms && receipt.run.arms[mode]) || {};
206
+ arms[mode] = { model_id: a.model_id || modelId, generated_at: a.generated_at || receipt.run.generated_at || receipt.run.date_utc };
207
+ const perCase = modeCases.map((c) => ({
208
+ c,
209
+ counts: {
210
+ ...(c.generation && Array.isArray(c.generation.draws) ? { generations_per_arm: c.generation.draws.length } : {}),
211
+ ...(!c.case_status ? { judge_samples_per_generation: samples } : {}),
212
+ },
213
+ }));
214
+ for (const kind of ['generations_per_arm', 'judge_samples_per_generation']) {
215
+ const values = perCase.map((x) => (Number.isInteger(x.counts[kind]) ? x.counts[kind] : null));
216
+ if (values.every((v) => v !== null) && new Set(values).size === 1) arms[mode].counts = { ...(arms[mode].counts || {}), [kind]: values[0] };
217
+ else for (const x of perCase) if (Number.isInteger(x.counts[kind])) x.c.counts = { ...(x.c.counts || {}), [kind]: x.counts[kind] };
218
+ }
219
+ }
220
+ const graderRevision = {
221
+ prompt_template_hash: judgeBlock.prompt_template_hash,
222
+ rubric_hashes: cases.map((c) => { const k = skill.suite.cases.find((x) => x.id === c.id); return k ? rubricHash(k.rubric) : null; }),
223
+ };
224
+ const out = buildReceipt({
225
+ skill: { name: receipt.skill.name, version: receipt.skill.version, contentHash: receipt.skill.content_hash, tokens: receipt.skill.tokens },
226
+ suite: { format: receipt.suite.format, suiteHash: receipt.suite.suite_hash, caseCount: receipt.suite.case_count, canary: receipt.suite.canary },
227
+ run: {
228
+ model_id: modelId,
229
+ model_release_date: receipt.run.model_release_date,
230
+ provider: receipt.run.provider,
231
+ surface,
232
+ surface_overhead_note: receipt.run.surface_overhead_note,
233
+ runner_version: RUNNER_VERSION,
234
+ date_utc: nowIso,
235
+ registry: receipt.run.registry,
236
+ transcripts: 'hashes-only',
237
+ judge: judgeBlock,
238
+ pricing_snapshot: pricingSnapshot,
239
+ answered_by: receipt.run.answered_by,
240
+ arms,
241
+ judged_at: judgedAt,
242
+ grader_revision: graderRevision,
243
+ },
244
+ cases,
245
+ economics,
246
+ verificationLevel: receipt.verification_level === 'TESTED' && judgeModelAnswered ? 'TESTED' : 'UNVERIFIED',
247
+ });
248
+
249
+ // Where the regrade came from: what v0.8 has no field for. The judge time, the grader
250
+ // revision and the archived arms are in the receipt (spec 043 AC-4) and are not repeated
251
+ // here; this sidecar is bound to the receipt by its hash.
252
+ const provenance = {
253
+ format: 'driftproof-regrade/2',
254
+ receipt_hash: out.receipt_hash,
255
+ regraded_from: { receipt_hash: receipt.receipt_hash, runner_version: receipt.run.runner_version, date_utc: receipt.run.date_utc, judge: receipt.run.judge },
256
+ generated_at_basis: receipt.run.generated_at || (receipt.run.arms && Object.values(receipt.run.arms).some((x) => x && x.generated_at))
257
+ ? "the original receipt's own generated_at"
258
+ : "the original receipt's run.date_utc: a receipt of that schema records no separate generation time",
259
+ draws: { graded: drawsToGrade(receipt).length, carried: receipt.results.cases.reduce((a, c) => a + c.generation.draws.length, 0) - drawsToGrade(receipt).length },
260
+ judge_reported_models: [...new Set(replies.flatMap((r) => (r.reportedModels || []).map((id) => canonicalModelId(id))))].sort(),
261
+ calls,
262
+ };
263
+ return { receipt: out, provenance, calls };
264
+ }
265
+
266
+ module.exports = { regradeReceipt, planRegrade, answerProblems, drawsToGrade, gradable };