driftproof 0.11.1 → 0.11.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/decision.js CHANGED
@@ -21,6 +21,7 @@ const fs = require('fs');
21
21
  const path = require('path');
22
22
  const { EFFECT_FLOOR } = require('../config');
23
23
  const { shortModel, githubOutputEntry, receiptVerdict, drawsLine, UNDERPOWERED_LINE } = require('./verdict');
24
+ const { duplicateCaseRows, ambiguityLine } = require('./receipt');
24
25
 
25
26
  // The six decision states (spec 030 AC-4).
26
27
  //
@@ -86,6 +87,12 @@ function failsJob(state, { failOnRegression = true } = {}) {
86
87
  // the fail-safe direction.
87
88
  function decisionState(receipt) {
88
89
  if (!receipt) return 'refused';
90
+ // Spec 050 AC-3: a receipt with two rows for one (id, mode) is `refused` before
91
+ // anything else is read. It is not `not measured`, which does not fail a job: the
92
+ // receipt may well have measured a regression, and which row a reader keeps decides
93
+ // whether that regression is seen. `refused` fails closed whatever
94
+ // fail-on-regression says, which is the direction an ambiguity has to fail.
95
+ if (duplicateCaseRows(receipt).length) return 'refused';
89
96
  const level = receipt.verification_level;
90
97
  const kind = receipt.run && receipt.run.answered_by ? receipt.run.answered_by.kind : undefined;
91
98
  if (level !== 'TESTED' || kind !== 'model') return 'not measured';
@@ -174,29 +181,45 @@ function decideSet(dir, requestedCsv, { failOnRegression = true } = {}) {
174
181
  });
175
182
  const used = new Set();
176
183
  const rows = requested.map((model) => {
177
- const hit = loaded.find((l) => !used.has(l.file) && l.receipt
184
+ // Spec 050 AC-6: EVERY readable receipt that matches, not the first. Until this the
185
+ // row took `loaded.find(...)`, the first match in sorted filename order, and reported
186
+ // the rest as unexpected. A job with two Driftproof steps shared one directory, so
187
+ // the second step's model row was filled by whichever skill's receipt sorted first
188
+ // (an outside correctness audit, finding 2; spec 030's F-1). Two receipts for one
189
+ // model are a question nobody answered, so the row is `refused`, names every file,
190
+ // and reads none of them.
191
+ const matches = loaded.filter((l) => !used.has(l.file) && l.receipt
178
192
  && l.receipt.run && matchesModel(l.receipt.run.model_id, model));
179
- if (hit) used.add(hit.file);
193
+ for (const m of matches) used.add(m.file);
194
+ const many = matches.length > 1;
195
+ const hit = matches.length === 1 ? matches[0] : null;
180
196
  // A receipt that exists and cannot be parsed is UNREADABLE, not absent.
181
197
  // Both fail closed, and the row says which - the register's
182
198
  // absence-vs-unreadable distinction, kept at the point it is decided.
183
- const unreadable = !hit && loaded.find((l) => !used.has(l.file) && l.error
199
+ const unreadable = !hit && !many && loaded.find((l) => !used.has(l.file) && l.error
184
200
  && fileNamesModel(l.file, model));
185
201
  if (unreadable) used.add(unreadable.file);
186
202
  const receipt = hit ? hit.receipt : null;
203
+ // Spec 050 AC-3: one receipt, but one with two rows for a case and arm.
204
+ const dups = receipt ? duplicateCaseRows(receipt) : [];
205
+ const ambiguous = many
206
+ ? `${matches.length} receipts for this model (${matches.map((m) => path.basename(m.file)).join(', ')}); none was read`
207
+ : (dups.length ? `receipt ambiguous: ${ambiguityLine(dups)}` : null);
187
208
  const state = decisionState(receipt);
188
209
  return {
189
210
  model,
190
211
  state,
191
- delta: receipt && receipt.comparison && typeof receipt.comparison.delta === 'number'
212
+ // An ambiguous row carries no delta: a delta is a reading, and none was taken.
213
+ delta: !ambiguous && receipt && receipt.comparison && typeof receipt.comparison.delta === 'number'
192
214
  ? receipt.comparison.delta : null,
193
- file: hit ? path.basename(hit.file) : (unreadable ? path.basename(unreadable.file) : null),
215
+ file: many ? null : hit ? path.basename(hit.file) : (unreadable ? path.basename(unreadable.file) : null),
194
216
  // Spec 035 AC-3: what an underpowered row would have needed, from the same reading.
195
217
  drawsNeeded: state === 'underpowered' ? receiptVerdict(receipt).drawsNeeded : null,
196
218
  // The distinction is carried on the ROW, not left in a sentence, because
197
219
  // the enforcement message has to say which of the two happened (AC-2's
198
220
  // mutation class, absence-vs-unreadable).
199
221
  unreadable: unreadable ? unreadable.error : null,
222
+ ambiguous,
200
223
  // A SHORT CAUSE, not the parser's text. V8's JSON.parse message echoes the
201
224
  // first bytes of the input - "Unexpected token '|', \"|{bad\" is not valid
202
225
  // JSON" - and this string is rendered into a markdown table cell, where one
@@ -204,7 +227,7 @@ function decideSet(dir, requestedCsv, { failOnRegression = true } = {}) {
204
227
  // reaches the annotation, which is not a table; the cell carries the cause,
205
228
  // and the filename is already beside it in the same cell (F-2 of
206
229
  // 2026-09-12).
207
- reason: unreadable ? 'receipt unreadable (not valid JSON)' : (hit ? null : 'no receipt for this model'),
230
+ reason: ambiguous || (unreadable ? 'receipt unreadable (not valid JSON)' : (hit ? null : 'no receipt for this model')),
208
231
  };
209
232
  });
210
233
  // Receipts the run wrote for models nobody requested are reported rather than
@@ -226,9 +249,14 @@ function decideSet(dir, requestedCsv, { failOnRegression = true } = {}) {
226
249
  // split it by CAUSE: nothing was written for this model, or something was
227
250
  // written and cannot be read. Both fail closed; they are not the same fact,
228
251
  // and the failure has to say which.
229
- absent: rows.filter((r) => r.state === 'refused' && !r.unreadable).map((r) => r.model),
252
+ absent: rows.filter((r) => r.state === 'refused' && !r.unreadable && !r.ambiguous).map((r) => r.model),
230
253
  unreadable: rows.filter((r) => r.state === 'refused' && r.unreadable)
231
254
  .map((r) => ({ model: r.model, file: r.file, error: r.unreadable })),
255
+ // Spec 050: a third cause of `refused`, apart from the other two for the same reason
256
+ // they are apart from each other. Something was written, and it can be read, but it
257
+ // does not say one thing.
258
+ ambiguous: rows.filter((r) => r.state === 'refused' && r.ambiguous)
259
+ .map((r) => ({ model: r.model, reason: r.ambiguous })),
232
260
  unexpected,
233
261
  receiptCount: files.length,
234
262
  requestedCount: requested.length,
@@ -390,6 +418,11 @@ function enforcementLines(d, { failOnRegression = true } = {}) {
390
418
  + `${d.unreadable.map((u) => `${u.model} exists and is unreadable (${u.file}: ${oneLine(u.error)})`).join('; ')}. `
391
419
  + `Failing closed: a receipt that cannot be read is not a pass, and is not the same as one that was never written.`);
392
420
  }
421
+ if (d.ambiguous && d.ambiguous.length) {
422
+ L.push(`::error title=Driftproof::The receipt set is ambiguous for `
423
+ + `${d.ambiguous.map((a) => `${a.model}: ${oneLine(a.reason, 240)}`).join('; ')}. `
424
+ + `Failing closed: a decision that depends on which receipt or row was read is not a pass.`);
425
+ }
393
426
  if (d.regressed.length) {
394
427
  const deltas = d.rows.filter((r) => r.state === 'regression')
395
428
  .map((r) => `${r.model} (delta ${r.delta})`).join(', ');
package/lib/diff.js CHANGED
@@ -1,6 +1,7 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
+ const { judgeSamplesOrNull } = require('./counts');
4
5
  const { bandVerdict, round, WITHIN_NOISE } = require('./stats');
5
6
  const { EFFECT_FLOOR } = require('../config');
6
7
  const { revisionHeadline } = require('./revision');
@@ -158,6 +159,12 @@ function revisionPairProblem(a, b) {
158
159
  return null;
159
160
  }
160
161
 
162
+ // The run's judge-sample count as a reader shows it: the number, or 'unknown'.
163
+ function judgeN(r) {
164
+ const n = judgeSamplesOrNull(r);
165
+ return n === null ? 'unknown' : n;
166
+ }
167
+
161
168
  function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' } = {}) {
162
169
  const revision = mode === 'revision';
163
170
 
@@ -253,7 +260,7 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
253
260
  L.push(`| surface (held) | ${a.run.surface} | ${b.run.surface} |`);
254
261
  L.push(`| suite_hash (held) | \`${short(a.suite.suite_hash)}\` | \`${short(b.suite.suite_hash)}\` |`);
255
262
  L.push(`| run date (UTC) | ${a.run.date_utc} | ${b.run.date_utc} |`);
256
- L.push(`| judge samples/case | ${(a.run.judge || {}).samples || 1} | ${(b.run.judge || {}).samples || 1} |`);
263
+ L.push(`| judge samples/case | ${judgeN(a)} | ${judgeN(b)} |`);
257
264
  L.push(`| with_skill (mean ± band) | ${bandStr(aAgg)} | ${bandStr(bAgg)} |`);
258
265
  L.push(`| baseline score | ${pct(a.comparison.baseline_score)} | ${pct(b.comparison.baseline_score)} |`);
259
266
  L.push(`| skill lift (Δ) | ${fmt(a.comparison.delta)} | ${fmt(b.comparison.delta)} |`);
@@ -268,7 +275,7 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
268
275
  L.push(`| model | \`${a.run.model_id}\` | \`${b.run.model_id}\` |`);
269
276
  L.push(`| run date (UTC) | ${a.run.date_utc} | ${b.run.date_utc} |`);
270
277
  L.push(`| surface | ${a.run.surface} | ${b.run.surface} |`);
271
- L.push(`| judge samples/case | ${(a.run.judge || {}).samples || 1} | ${(b.run.judge || {}).samples || 1} |`);
278
+ L.push(`| judge samples/case | ${judgeN(a)} | ${judgeN(b)} |`);
272
279
  L.push(`| skill content_hash | \`${short(a.skill.content_hash)}\` | \`${short(b.skill.content_hash)}\` |`);
273
280
  L.push(`| suite_hash | \`${short(a.suite.suite_hash)}\` | \`${short(b.suite.suite_hash)}\` |`);
274
281
  L.push(`| with_skill (mean ± band) | ${bandStr(aAgg)} | ${bandStr(bAgg)} |`);
@@ -289,7 +296,9 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
289
296
  if (!revision && a.skill.content_hash !== b.skill.content_hash) warnings.push('skill content_hash differs — the skill itself changed between receipts, so drift mixes skill edits with model drift.');
290
297
  if (!revision && a.suite.suite_hash !== b.suite.suite_hash) warnings.push('suite_hash differs — the eval suite changed; per-case comparison may be misleading.');
291
298
  if (a.skill.name !== b.skill.name) warnings.push(`different skills (${a.skill.name} vs ${b.skill.name}) — comparison is not meaningful.`);
292
- if ((a.run.judge || {}).samples <= 1 || (b.run.judge || {}).samples <= 1) warnings.push('one or both receipts are single-sample (no bands) — non-overlap can only be trusted when both sides are sampled.');
299
+ // v0.8 (spec 043 AC-1): an absent judge count is unknown, said as such, never read as 1.
300
+ if (judgeN(a) === 'unknown' || judgeN(b) === 'unknown') warnings.push('the judge-sample count is unknown on one or both receipts: the receipt does not establish it, so whether the bands are sampled cannot be read from it.');
301
+ if (judgeN(a) === 1 || judgeN(b) === 1) warnings.push('one or both receipts are single-sample (no bands): non-overlap can only be trusted when both sides are sampled.');
293
302
  // WHAT SUPPRESSED THE VERDICT IS SAID, AND IT IS SAID CORRECTLY (spec 016 AC-3
294
303
  // and AC-4, closing F-015-B).
295
304
  //
package/lib/export.js CHANGED
@@ -15,6 +15,8 @@ const SUMMARY_FORMAT_VERSION = '1';
15
15
  // Build the summary object for one receipt. Deterministic for a given receipt:
16
16
  // fixed key order, no export-time timestamps. `reportUrl` is caller-supplied
17
17
  // (receipts do not know where their report lives), else null.
18
+ const { judgeSamplesOrNull } = require('./counts');
19
+
18
20
  function toSummaryJson(receipt, { reportUrl = null } = {}) {
19
21
  const agg = receipt.results.aggregates;
20
22
  const cmp = receipt.comparison || {};
@@ -41,7 +43,8 @@ function toSummaryJson(receipt, { reportUrl = null } = {}) {
41
43
  source: receipt.run.source || 'driftproof',
42
44
  judge: {
43
45
  model_id: (receipt.results.cases.find((c) => c.judge) || { judge: { model_id: 'unknown' } }).judge.model_id,
44
- samples: (receipt.run.judge || {}).samples || 1,
46
+ // v0.8 (spec 043 AC-1): null when the receipt does not establish the count, never 1.
47
+ samples: judgeSamplesOrNull(receipt),
45
48
  },
46
49
  receipt_hash: receipt.receipt_hash,
47
50
  report_url: reportUrl,
@@ -0,0 +1,451 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // Receipt interop: the two Anthropic formats (spec 049, receipt spec v0.9).
5
+ //
6
+ // claude-plugin-eval `claude plugin eval`'s aggregate-result.json (or its --json output),
7
+ // schemaVersion 1. The unit measured is a plugin.
8
+ // skill-creator skill-creator's benchmark.json, which skill-up also writes, with
9
+ // skill-up's result.json read from beside it when present.
10
+ //
11
+ // The rules every importer here keeps (docs/interop.md): no source field is invented, so an
12
+ // absent one is "unknown" or null and named in run.import.notices; the receipt is DECLARED,
13
+ // surface external, every hash null; the run date comes from the source and never from the
14
+ // clock, which is read once, into run.import.imported_at; cases are grouped by their identity
15
+ // in the source, never by a display name. Nothing the source carries as text (prompts,
16
+ // answers, grader evidence, local paths) is copied.
17
+
18
+ const fs = require('fs');
19
+ const path = require('path');
20
+ const crypto = require('crypto');
21
+ const { RECEIPT_SCHEMA_VERSION, RUNNER_VERSION } = require('../config');
22
+ const { sealReceipt, BAND_RULE, comparisonOf } = require('./receipt');
23
+ const { mean, stddev, round, aggregateBands } = require('./stats');
24
+ const { outcomeFor } = require('./run');
25
+ const { inferProvider } = require('./provider');
26
+ const { registryStatus } = require('./models');
27
+ const { placeCounts } = require('./counts');
28
+ const { median, quartiles } = require('./value');
29
+
30
+ const ANTHROPIC_TOOLS = ['claude-plugin-eval', 'skill-creator'];
31
+ const DOCUMENT_NAME = { 'claude-plugin-eval': 'aggregate-result.json', 'skill-creator': 'benchmark.json' };
32
+ const AGGREGATES_ONLY = 'aggregates only: the source carries no per-run scores for this case, so no band can be formed';
33
+
34
+ class ImportRefused extends Error {}
35
+ const refuse = (msg) => { throw new ImportRefused(msg); };
36
+ const sha256 = (buf) => crypto.createHash('sha256').update(buf).digest('hex');
37
+ // Source text copied into a receipt (a run's error) can name local files: a home directory, a
38
+ // scratch directory, a user's project. Every local path is replaced before it is written (spec 049
39
+ // NFR-3, approval F-1 twice): a file: URL, a drive path with either slash, and an absolute or
40
+ // home-relative path wherever it starts, a colon before it included; an apostrophe followed by a
41
+ // word character is part of the path (A-052-2). Web URLs are set aside first
42
+ // and put back, so they are the one thing a slash-led run is not redacted inside. A path that starts
43
+ // right after a quote runs to the same quote, or to the end of the text, spaces and line breaks
44
+ // included (spec 052 AC-1 and A-052-1, spec 049 approval F-1 a third time): Node quotes the paths it
45
+ // prints, a profile folder can have a space in it, and a run error can be a stack of several lines.
46
+ const WEB_URL = /\b(?:https?|wss?|ftp):\/\/[^\s"'`<>]+/gi;
47
+ const PATH_START = /^(?:file:\/\/|[A-Za-z]:[\\/]|~?\/)/;
48
+ // A quoted path closes at the next same quote that is not inside a name (followed by a word
49
+ // character, as in O'Brien, A-052-2), across line breaks, unless that quote itself opens a
50
+ // path: then this one ends at its own line break (A-052-1, approval F-1: an unclosed quote in one
51
+ // stack line must not pair with the quote that opens the next line's path). With no such quote, the
52
+ // path runs to the end of the text.
53
+ function redactQuoted(s) {
54
+ let out = '';
55
+ let i = 0;
56
+ while (i < s.length) {
57
+ const q = s[i];
58
+ if ((q === "'" || q === '"' || q === '`') && PATH_START.test(s.slice(i + 1))) {
59
+ // A same quote followed by a word character is inside a name (O'Brien), not a closing quote,
60
+ // unless it opens a path of its own (A-052-2).
61
+ let k = s.indexOf(q, i + 1);
62
+ while (k >= 0 && /\w/.test(s[k + 1] || '') && !PATH_START.test(s.slice(k + 1))) k = s.indexOf(q, k + 1);
63
+ if (k < 0) { out += `${q}<local path>`; i = s.length; continue; }
64
+ if (!PATH_START.test(s.slice(k + 1))) { out += `${q}<local path>${q}`; i = k + 1; continue; }
65
+ const nl = s.indexOf('\n', i + 1);
66
+ const end = nl >= 0 && nl < k ? nl : k;
67
+ out += `${q}<local path>`; i = end; continue;
68
+ }
69
+ out += q; i++;
70
+ }
71
+ return out;
72
+ }
73
+ const LOCAL_PATHS = [
74
+ /\bfile:\/\/(?:[^\s"'`<>]|'(?=\w))*/gi,
75
+ /(?<![\w])[A-Za-z]:[\\/](?:[^\s"'`<>]|'(?=\w))*/g,
76
+ /(?<![\w.~-])~?\/(?:[^\s"'`<>]|'(?=\w))*[^\s"'`<>.,;:)\]]/g,
77
+ ];
78
+ function redactPaths(s) {
79
+ const urls = [];
80
+ let out = String(s).replace(WEB_URL, (u) => { urls.push(u); return `\u0000${urls.length - 1}\u0000`; });
81
+ out = redactQuoted(out);
82
+ for (const re of LOCAL_PATHS) out = out.replace(re, '<local path>');
83
+ return out.replace(/\u0000(\d+)\u0000/g, (_, n) => urls[Number(n)]);
84
+ }
85
+ // The last word on NFR-3: a receipt that still names a home directory is not written. A home
86
+ // directory is read wherever it sits in a path, not only at its start (spec 052 AC-2, spec 049
87
+ // approval F-2): /var/home and /usr/home, /mnt/c/Users and /System/Volumes/Data/Users, /var/root;
88
+ // a profile root after a drive or a WSL drive mount in either case; and home, Users or root between
89
+ // backslashes (a \\wsl$ share, a UNC path). The receipt is read as JSON, so a backslash is doubled.
90
+ // Web URLs are set aside first, as the redaction does.
91
+ const HOME_PATHS = [
92
+ /\/(?:home|Users)\/[^/\s"]+/,
93
+ /\/root\//,
94
+ /(?:[A-Za-z]:|\/mnt\/[A-Za-z])[\\/]+(?:users|documents and settings)[\\/]/i,
95
+ /\\(?:home|users|root)\\+[^\\\s"]/i,
96
+ ];
97
+ const namesHome = (text) => { const t = String(text).replace(WEB_URL, ''); return HOME_PATHS.some((re) => re.test(t)); };
98
+ const num = (x) => typeof x === 'number' && Number.isFinite(x);
99
+
100
+ // ── finding the document ─────────────────────────────────────────────────────
101
+ // A file is the document. A directory is searched, all the way down, for the one file the
102
+ // format needs; none, or more than one, is a refusal that names what was found.
103
+ function findDocument(p, from) {
104
+ const name = DOCUMENT_NAME[from];
105
+ if (!name) refuse(`unknown import source "${from}"`);
106
+ let st;
107
+ try { st = fs.statSync(p); } catch (_e) { refuse(`no such file or directory: ${p}`); }
108
+ if (st.isFile()) return p;
109
+ const found = [];
110
+ const walk = (d) => {
111
+ for (const e of fs.readdirSync(d, { withFileTypes: true }).sort((a, b) => (a.name < b.name ? -1 : 1))) {
112
+ const f = path.join(d, e.name);
113
+ if (e.isDirectory() && !e.isSymbolicLink()) walk(f);
114
+ else if (e.isFile() && e.name === name) found.push(f);
115
+ }
116
+ };
117
+ walk(p);
118
+ if (found.length === 1) return found[0];
119
+ if (!found.length) refuse(`no ${name} under ${p}: --from ${from} imports one ${name}`);
120
+ refuse(`${found.length} files named ${name} under ${p}; import one at a time:\n ${found.map((f) => path.relative(p, f)).join('\n ')}`);
121
+ }
122
+
123
+ // ── shared pieces ────────────────────────────────────────────────────────────
124
+ // One case row from its measured draws. `threshold` is the source's pass threshold, or null.
125
+ function measuredRow(id, mode, scores, threshold, judgeModel) {
126
+ const m = round(mean(scores));
127
+ const sd = round(stddev(scores));
128
+ return { id, mode, outcome: outcomeFor(m, sd, threshold), score: m, mean: m, stddev: sd, samples: scores.map((s) => round(s)), threshold, judge: { model_id: judgeModel, rubric_hash: null } };
129
+ }
130
+ function unobservedRow(id, mode) { return { id, mode, case_status: 'no_observations' }; }
131
+
132
+ function aggregateMode(rows) {
133
+ const band = aggregateBands(rows.map((c) => ({ mean: c.mean, stddev: c.stddev || 0, n: c.samples.length })));
134
+ return {
135
+ case_count: rows.length,
136
+ pass_count: rows.filter((c) => c.outcome === 'pass').length,
137
+ borderline_count: rows.filter((c) => c.outcome === 'borderline').length,
138
+ mean_score: band.mean,
139
+ stddev: band.stddev,
140
+ };
141
+ }
142
+
143
+ // Economics per arm, from what the source reports per run. `costs` are the source's own
144
+ // estimates (Format A) or absent (Format B); `wallMs` and `tokens` as reported.
145
+ function armEconomics({ n, costs = [], wallMs = [], tokens = null }) {
146
+ const q = quartiles(wallMs);
147
+ const out = {
148
+ call_count: n,
149
+ mean_input_tokens: null,
150
+ mean_output_tokens: null,
151
+ mean_cost_usd_per_call: costs.length ? round(mean(costs)) : null,
152
+ median_wall_ms: median(wallMs),
153
+ wall_ms_p25: q.p25,
154
+ wall_ms_p75: q.p75,
155
+ wall_ms_iqr: q.iqr,
156
+ };
157
+ if (tokens) out.mean_total_tokens = tokens.length ? round(mean(tokens), 2) : null;
158
+ return out;
159
+ }
160
+ function economicsBlock(basis, w, b) {
161
+ const inc = w.mean_cost_usd_per_call != null && b.mean_cost_usd_per_call != null ? round(w.mean_cost_usd_per_call - b.mean_cost_usd_per_call) : null;
162
+ return {
163
+ basis,
164
+ with_skill: w,
165
+ baseline: b,
166
+ skill_incremental_cost_usd_per_call: inc,
167
+ skill_incremental_cost_usd_per_1k_calls: inc == null ? null : round(inc * 1000),
168
+ output_tokens_delta: null,
169
+ median_wall_ms_delta: w.median_wall_ms != null && b.median_wall_ms != null ? round(w.median_wall_ms - b.median_wall_ms, 2) : null,
170
+ judge_excluded: true,
171
+ };
172
+ }
173
+
174
+ // The receipt shell. `rows` are results.cases in order; `perCase` the counts each row's
175
+ // source establishes; `excludedReasons` the reason per excluded row index.
176
+ function assemble({ tool, format, formatVersion, sourceBytes, sidecars, importedAt, notices, skill, suite, modelId, judgeModel, dateUtc, harness, rows, perCase, excludedReasons, economics }) {
177
+ const measured = rows.filter((c) => !c.case_status);
178
+ const aggW = aggregateMode(measured.filter((c) => c.mode === 'with_skill'));
179
+ const aggB = aggregateMode(measured.filter((c) => c.mode === 'baseline'));
180
+ const excluded = rows.map((c, i) => [c, i]).filter(([c]) => c.case_status).map(([c, i]) => ({ id: c.id, modes: [c.mode], reason: excludedReasons[i] }));
181
+ const imp = { tool, format, format_version: formatVersion, source_sha256: sha256(sourceBytes), imported_at: importedAt };
182
+ if (sidecars.length) imp.sidecars = sidecars;
183
+ if (notices.length) imp.notices = notices;
184
+ const receipt = {
185
+ schema_version: RECEIPT_SCHEMA_VERSION,
186
+ skill: { name: skill.name, version: skill.version, content_hash: null, unit: skill.unit },
187
+ suite: { format: suite.format, suite_hash: null, case_count: suite.caseCount },
188
+ run: {
189
+ model_id: modelId,
190
+ model_release_date: null,
191
+ provider: modelId === 'unknown' ? 'unknown' : inferProvider(modelId),
192
+ surface: 'external',
193
+ source: `imported/${tool}`,
194
+ runner_version: RUNNER_VERSION,
195
+ date_utc: dateUtc,
196
+ registry: registryStatus(modelId),
197
+ transcripts: 'none',
198
+ judge: { temperature: null, sampling: 'external', surface: 'external', model_id: judgeModel, prompt_template_hash: null },
199
+ answered_by: { kind: 'external', attested: false, reported_model: null, reported_models: null, isolation: 'none' },
200
+ ...(harness ? { harness } : {}),
201
+ import: imp,
202
+ },
203
+ results: {
204
+ cases: rows,
205
+ aggregates: { with_skill: aggW, baseline: aggB, band_rule: BAND_RULE, ...(excluded.length ? { excluded_cases: excluded } : {}) },
206
+ },
207
+ comparison: comparisonOf(aggW, aggB),
208
+ verification_level: 'DECLARED',
209
+ economics,
210
+ receipt_hash: '',
211
+ };
212
+ placeCounts(receipt, perCase);
213
+ if (namesHome(JSON.stringify(receipt))) refuse('the receipt would carry a home-directory path copied from the source; nothing is written (spec 049 NFR-3)');
214
+ return sealReceipt(receipt);
215
+ }
216
+
217
+ // ── Format A: claude plugin eval ─────────────────────────────────────────────
218
+ // The results directory's name, `2026-09-10T17-02-11-482Z`, is a UTC time.
219
+ function dateFromDirName(file) {
220
+ const m = /^(\d{4}-\d{2}-\d{2})T(\d{2})-(\d{2})-(\d{2})(?:-(\d{3}))?Z$/.exec(path.basename(path.dirname(file)));
221
+ return m ? `${m[1]}T${m[2]}:${m[3]}:${m[4]}${m[5] ? `.${m[5]}` : ''}Z` : null;
222
+ }
223
+
224
+ function importPluginEval(file, bytes, { importedAt, countErroredRuns = false }) {
225
+ let d;
226
+ try { d = JSON.parse(bytes.toString('utf8')); } catch (e) { refuse(`not JSON: ${e.message}`); }
227
+ if (!d || typeof d !== 'object' || Array.isArray(d)) refuse('expected a claude plugin eval result object');
228
+ if (d.schemaVersion !== 1) refuse(`claude plugin eval schemaVersion ${JSON.stringify(d.schemaVersion)} is not supported (this importer reads schemaVersion 1)`);
229
+ if (d.partial === true) refuse(`the document is partial, and a partial run is not imported: ${d.partialReason == null ? 'the source gives no partialReason' : String(d.partialReason)}`);
230
+ if (!Array.isArray(d.cases)) refuse('the document carries no cases[]');
231
+ const names = d.cases.map((c) => (c && typeof c.name === 'string' ? c.name : null));
232
+ if (names.some((n) => n === null)) refuse('a case carries no name');
233
+ const dup = names.find((n, i) => names.indexOf(n) !== i);
234
+ if (dup !== undefined) refuse(`two cases share the name ${JSON.stringify(dup)}; cases are never merged`);
235
+
236
+ const suite = d.suite || {};
237
+ const notices = [];
238
+ const modelId = typeof suite.modelOverride === 'string' && suite.modelOverride ? suite.modelOverride : 'unknown';
239
+ if (modelId === 'unknown') notices.push('the source records no model under test (suite.modelOverride is absent): run.model_id is unknown, never Claude Code\'s default');
240
+ const judgeModel = typeof suite.judgeModel === 'string' && suite.judgeModel ? suite.judgeModel : 'unknown';
241
+ if (judgeModel === 'unknown') notices.push('the source records no judge model (suite.judgeModel is absent); its default judge is a small fast model, so run.judge.model_id is unknown');
242
+ let dateUtc = typeof d.startedAt === 'string' ? d.startedAt : null;
243
+ if (!dateUtc) {
244
+ dateUtc = dateFromDirName(file);
245
+ notices.push(dateUtc ? 'run.date_utc is read from the results directory\'s name (the document carries no startedAt)' : 'the source carries no run date (no startedAt, and the directory name is not a results timestamp): run.date_utc is null');
246
+ }
247
+ const plugins = Array.isArray(suite.plugins) ? suite.plugins : [];
248
+ const plugin = plugins.length === 1 ? plugins[0] : null;
249
+ if (!plugin) notices.push(`the suite loads ${plugins.length} plugins, not one: skill.name and skill.version are unknown`);
250
+ const harness = { name: 'claude-code', version: typeof d.claudeVersion === 'string' ? d.claudeVersion : null };
251
+ if (harness.version === null) notices.push('the source records no claudeVersion: run.harness.version is null');
252
+ const threshold = num(suite.threshold) ? suite.threshold : null;
253
+ const twoArm = suite.ablation !== 'none';
254
+ let erroredCounted = 0;
255
+
256
+ const rows = [];
257
+ const perCase = [];
258
+ const excludedReasons = {};
259
+ const econ = { with_skill: { n: 0, costs: [], wallMs: [] }, baseline: { n: 0, costs: [], wallMs: [] } };
260
+ for (const c of d.cases) {
261
+ const arms = c.arms && typeof c.arms === 'object' ? c.arms : null;
262
+ const hasRuns = !!arms && (Array.isArray(arms.with) || Array.isArray(arms.without));
263
+ const skillGraders = (Array.isArray(c.graders) ? c.graders : []).filter((g) => g && g.type === 'tool_used' && g.config && g.config.tool === 'Skill').map((g) => g.name);
264
+ for (const [key, mode] of [['with', 'with_skill'], ['without', 'baseline']]) {
265
+ if (mode === 'baseline' && !(hasRuns ? Array.isArray(arms.without) : twoArm)) continue;
266
+ const idx = rows.length;
267
+ if (!hasRuns) {
268
+ rows.push(unobservedRow(c.name, mode));
269
+ excludedReasons[idx] = AGGREGATES_ONLY;
270
+ continue;
271
+ }
272
+ const runs = Array.isArray(arms[key]) ? arms[key] : [];
273
+ const measured = [];
274
+ const excludedDraws = [];
275
+ runs.forEach((r, i) => {
276
+ const at = i + 1;
277
+ if (r && r.error != null && !countErroredRuns) { excludedDraws.push({ draw_index: at, reason: `the source recorded an error for this run: ${redactPaths(r.error).slice(0, 300)}` }); return; }
278
+ if (r && r.skippedPaidGraders === true) { excludedDraws.push({ draw_index: at, reason: 'the source skipped this run\'s paid graders (skippedPaidGraders: true), so its score omits them' }); return; }
279
+ if (!r || !num(r.score)) { excludedDraws.push({ draw_index: at, reason: 'the source gives this run no numeric score' }); return; }
280
+ if (r.error != null) erroredCounted++;
281
+ measured.push(r);
282
+ });
283
+ perCase.push({ index: idx, counts: { generations_per_arm: measured.length } });
284
+ if (!measured.length) {
285
+ const row = unobservedRow(c.name, mode);
286
+ if (excludedDraws.length) row.excluded_draws = excludedDraws;
287
+ rows.push(row);
288
+ excludedReasons[idx] = runs.length ? 'every run of this arm was kept out; each is listed in excluded_draws' : 'the source lists no runs for this arm';
289
+ continue;
290
+ }
291
+ const row = measuredRow(c.name, mode, measured.map((r) => r.score), threshold, judgeModel);
292
+ if (excludedDraws.length) row.excluded_draws = excludedDraws;
293
+ // Activation: an unscored tool_used Skill grader says whether the skill fired. It is
294
+ // recorded beside the score and never enters it (the run's score already excludes it).
295
+ if (mode === 'with_skill') {
296
+ const activation = [];
297
+ for (const name of skillGraders) {
298
+ const seen = measured.map((r) => (Array.isArray(r.graders) ? r.graders.find((g) => g && g.name === name) : null)).filter((g) => g && g.scored === false);
299
+ if (seen.length) activation.push({ indicator: name, fired: seen.filter((g) => g.passed === true).length, runs: seen.length });
300
+ }
301
+ if (activation.length) row.activation = activation;
302
+ }
303
+ rows.push(row);
304
+ econ[mode].n += measured.length;
305
+ for (const r of measured) {
306
+ if (num(r.costUsd)) econ[mode].costs.push(r.costUsd);
307
+ if (num(r.durationSeconds)) econ[mode].wallMs.push(Math.round(r.durationSeconds * 1000));
308
+ }
309
+ }
310
+ }
311
+ if (erroredCounted) notices.push(`${erroredCounted} errored run(s) are counted as measured draws (--count-errored-runs)`);
312
+ return assemble({
313
+ tool: 'claude-plugin-eval', format: 'aggregate-result.json', formatVersion: d.schemaVersion, sourceBytes: bytes, sidecars: [], importedAt, notices,
314
+ skill: { name: plugin && typeof plugin.name === 'string' ? plugin.name : 'unknown', version: plugin && typeof plugin.version === 'string' ? plugin.version : 'unknown', unit: 'plugin' },
315
+ suite: { format: 'claude-plugin-eval/aggregate-result.json', caseCount: d.cases.length },
316
+ modelId, judgeModel, dateUtc, harness, rows, perCase, excludedReasons,
317
+ economics: economicsBlock('source-list-price-estimate', armEconomics(econ.with_skill), armEconomics(econ.baseline)),
318
+ });
319
+ }
320
+
321
+ // ── Format B: benchmark.json (skill-creator, skill-up) ───────────────────────
322
+ const CONFIGURATIONS = { with_skill: 'with_skill', without_skill: 'baseline' };
323
+
324
+ // skill-up's result.json names the model it forwarded (applied_configuration.model) and,
325
+ // only when the agent reported one, the model observed. Observed first.
326
+ function modelFromResult(res) {
327
+ const oc = res && res.observed_configuration;
328
+ if (oc && typeof oc.model === 'string' && oc.model) return { model: oc.model, from: 'observed_configuration.model' };
329
+ const per = (Array.isArray(res && res.case_results) ? res.case_results : []).map((c) => c && c.observed_model);
330
+ if (per.length && per.every((m) => typeof m === 'string' && m && m === per[0])) return { model: per[0], from: 'case_results[].observed_model' };
331
+ const ac = res && res.applied_configuration;
332
+ if (ac && typeof ac.model === 'string' && ac.model) return { model: ac.model, from: 'applied_configuration.model' };
333
+ return null;
334
+ }
335
+
336
+ function importBenchmark(file, bytes, { importedAt }) {
337
+ let d;
338
+ try { d = JSON.parse(bytes.toString('utf8')); } catch (e) { refuse(`not JSON: ${e.message}`); }
339
+ if (!d || typeof d !== 'object' || !d.metadata || typeof d.metadata !== 'object') refuse('expected a benchmark.json with metadata');
340
+ const meta = d.metadata;
341
+ const runs = Array.isArray(d.runs) ? d.runs : [];
342
+ const notices = [];
343
+ const groups = new Map();
344
+ const labels = new Map();
345
+ const seen = new Set();
346
+ for (const r of runs) {
347
+ if (!r || (typeof r.eval_id !== 'number' && typeof r.eval_id !== 'string')) refuse('a runs[] row carries no eval_id');
348
+ const mode = CONFIGURATIONS[r.configuration];
349
+ if (!mode) refuse(`configuration ${JSON.stringify(r.configuration)} is neither with_skill nor without_skill`);
350
+ const id = String(r.eval_id);
351
+ const key = `${id}\u0000${r.configuration}\u0000${r.run_number}`;
352
+ if (seen.has(key)) refuse(`two rows share eval_id ${id}, configuration ${r.configuration} and run_number ${r.run_number}; rows are never merged`);
353
+ seen.add(key);
354
+ if (!groups.has(id)) groups.set(id, { with_skill: [], baseline: [] });
355
+ groups.get(id)[mode].push(r);
356
+ if (typeof r.eval_name === 'string' && !labels.has(id)) labels.set(id, r.eval_name);
357
+ }
358
+ // Evals the metadata names with no row are aggregates only.
359
+ const declared = (Array.isArray(meta.evals_run) ? meta.evals_run : []).map(String);
360
+ const ids = [...new Set([...groups.keys(), ...declared])];
361
+ const hasBaseline = runs.some((r) => r.configuration === 'without_skill') || !!(d.run_summary && d.run_summary.without_skill);
362
+
363
+ // The model, from the document or the result.json beside it.
364
+ const sidecars = [];
365
+ let modelId = typeof meta.executor_model === 'string' && meta.executor_model ? meta.executor_model : null;
366
+ let harness = null;
367
+ const resultFile = path.join(path.dirname(file), 'result.json');
368
+ let res = null;
369
+ if (fs.existsSync(resultFile)) {
370
+ const rb = fs.readFileSync(resultFile);
371
+ try { res = JSON.parse(rb.toString('utf8')); sidecars.push({ file: 'result.json', sha256: sha256(rb) }); } catch (_e) { notices.push('result.json beside the document does not parse and is not read'); }
372
+ }
373
+ if (!modelId) {
374
+ const found = res && modelFromResult(res);
375
+ if (found) {
376
+ modelId = found.model;
377
+ notices.push(found.from === 'applied_configuration.model'
378
+ ? 'run.model_id is the model skill-up forwarded to the agent (result.json applied_configuration.model); the agent reported none, so it is not an observation'
379
+ : `run.model_id is read from result.json ${found.from}`);
380
+ } else {
381
+ modelId = 'unknown';
382
+ notices.push(res ? 'neither benchmark.json nor result.json records the model: run.model_id is unknown' : 'benchmark.json records no executor_model and no result.json is beside it: run.model_id is unknown');
383
+ }
384
+ }
385
+ if (res && typeof res.engine_name === 'string' && res.engine_name) {
386
+ const v = res.observed_configuration && typeof res.observed_configuration.version === 'string' ? res.observed_configuration.version : null;
387
+ harness = { name: res.engine_name, version: v };
388
+ }
389
+ // No cited benchmark.json format records the grader's model (skill-up's grading.json carries
390
+ // expectations and a summary only), and analyzer_model is not the grader, so the judge is unknown.
391
+ const judgeModel = 'unknown';
392
+ notices.push('benchmark.json records no grader model: run.judge.model_id is unknown (analyzer_model is not the grader and is not read)');
393
+ const dateUtc = typeof meta.timestamp === 'string' && meta.timestamp ? meta.timestamp : null;
394
+ if (!dateUtc) notices.push('benchmark.json carries no metadata.timestamp: run.date_utc is null');
395
+
396
+ const rows = [];
397
+ const perCase = [];
398
+ const excludedReasons = {};
399
+ const econ = { with_skill: { n: 0, wallMs: [], tokens: [] }, baseline: { n: 0, wallMs: [], tokens: [] } };
400
+ const observed = new Set();
401
+ for (const id of ids) {
402
+ const g = groups.get(id);
403
+ for (const mode of ['with_skill', 'baseline']) {
404
+ if (mode === 'baseline' && !hasBaseline) continue;
405
+ const idx = rows.length;
406
+ const rs = g ? g[mode].slice().sort((a, b) => Number(a.run_number) - Number(b.run_number)) : [];
407
+ const scored = rs.filter((r) => r.result && num(r.result.pass_rate));
408
+ if (!g) {
409
+ rows.push(unobservedRow(id, mode));
410
+ excludedReasons[idx] = AGGREGATES_ONLY;
411
+ continue;
412
+ }
413
+ perCase.push({ index: idx, counts: { generations_per_arm: scored.length } });
414
+ observed.add(scored.length);
415
+ const row = scored.length ? measuredRow(id, mode, scored.map((r) => r.result.pass_rate), null, judgeModel) : unobservedRow(id, mode);
416
+ if (labels.has(id)) row.label = labels.get(id);
417
+ const dropped = rs.filter((r) => !scored.includes(r)).map((r) => ({ draw_index: Number(r.run_number), reason: 'the source gives this run no numeric pass_rate' }));
418
+ if (dropped.length) row.excluded_draws = dropped;
419
+ rows.push(row);
420
+ if (!scored.length) { excludedReasons[idx] = 'the source lists no scored run for this arm'; continue; }
421
+ econ[mode].n += scored.length;
422
+ for (const r of scored) {
423
+ if (num(r.result.time_seconds)) econ[mode].wallMs.push(Math.round(r.result.time_seconds * 1000));
424
+ if (num(r.result.tokens)) econ[mode].tokens.push(r.result.tokens);
425
+ }
426
+ }
427
+ }
428
+ if (num(meta.runs_per_configuration) && [...observed].some((n) => n !== meta.runs_per_configuration)) {
429
+ notices.push(`metadata.runs_per_configuration declares ${meta.runs_per_configuration}; the rows carry ${[...observed].sort().join(', ')} per configuration, and the rows are what is recorded`);
430
+ }
431
+ return assemble({
432
+ tool: 'skill-creator', format: 'benchmark.json', formatVersion: null, sourceBytes: bytes, sidecars, importedAt, notices,
433
+ skill: { name: typeof meta.skill_name === 'string' && meta.skill_name ? meta.skill_name : 'unknown', version: 'unknown', unit: 'skill' },
434
+ suite: { format: 'skill-creator/benchmark.json', caseCount: ids.length },
435
+ modelId, judgeModel, dateUtc, harness, rows, perCase, excludedReasons,
436
+ economics: economicsBlock('source-reported', armEconomics(econ.with_skill), armEconomics(econ.baseline)),
437
+ });
438
+ }
439
+
440
+ // Import one document. `p` is the file or directory the user named.
441
+ function importAnthropic(p, { from, importedAt, countErroredRuns = false } = {}) {
442
+ if (!ANTHROPIC_TOOLS.includes(from)) refuse(`unknown import source "${from}"`);
443
+ const file = findDocument(p, from);
444
+ const bytes = fs.readFileSync(file);
445
+ const receipt = from === 'claude-plugin-eval'
446
+ ? importPluginEval(file, bytes, { importedAt, countErroredRuns })
447
+ : importBenchmark(file, bytes, { importedAt });
448
+ return { receipt, file };
449
+ }
450
+
451
+ module.exports = { importAnthropic, findDocument, ANTHROPIC_TOOLS, DOCUMENT_NAME, ImportRefused };