driftproof 0.11.2 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,451 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // Receipt interop: the two Anthropic formats (spec 049, receipt spec v0.9).
5
+ //
6
+ // claude-plugin-eval `claude plugin eval`'s aggregate-result.json (or its --json output),
7
+ // schemaVersion 1. The unit measured is a plugin.
8
+ // skill-creator skill-creator's benchmark.json, which skill-up also writes, with
9
+ // skill-up's result.json read from beside it when present.
10
+ //
11
+ // The rules every importer here keeps (docs/interop.md): no source field is invented, so an
12
+ // absent one is "unknown" or null and named in run.import.notices; the receipt is DECLARED,
13
+ // surface external, every hash null; the run date comes from the source and never from the
14
+ // clock, which is read once, into run.import.imported_at; cases are grouped by their identity
15
+ // in the source, never by a display name. Nothing the source carries as text (prompts,
16
+ // answers, grader evidence, local paths) is copied.
17
+
18
+ const fs = require('fs');
19
+ const path = require('path');
20
+ const crypto = require('crypto');
21
+ const { RECEIPT_SCHEMA_VERSION, RUNNER_VERSION } = require('../config');
22
+ const { sealReceipt, BAND_RULE, comparisonOf } = require('./receipt');
23
+ const { mean, stddev, round, aggregateBands } = require('./stats');
24
+ const { outcomeFor } = require('./run');
25
+ const { inferProvider } = require('./provider');
26
+ const { registryStatus } = require('./models');
27
+ const { placeCounts } = require('./counts');
28
+ const { median, quartiles } = require('./value');
29
+
30
+ const ANTHROPIC_TOOLS = ['claude-plugin-eval', 'skill-creator'];
31
+ const DOCUMENT_NAME = { 'claude-plugin-eval': 'aggregate-result.json', 'skill-creator': 'benchmark.json' };
32
+ const AGGREGATES_ONLY = 'aggregates only: the source carries no per-run scores for this case, so no band can be formed';
33
+
34
+ class ImportRefused extends Error {}
35
+ const refuse = (msg) => { throw new ImportRefused(msg); };
36
+ const sha256 = (buf) => crypto.createHash('sha256').update(buf).digest('hex');
37
+ // Source text copied into a receipt (a run's error) can name local files: a home directory, a
38
+ // scratch directory, a user's project. Every local path is replaced before it is written (spec 049
39
+ // NFR-3, approval F-1 twice): a file: URL, a drive path with either slash, and an absolute or
40
+ // home-relative path wherever it starts, a colon before it included; an apostrophe followed by a
41
+ // word character is part of the path (A-052-2). Web URLs are set aside first
42
+ // and put back, so they are the one thing a slash-led run is not redacted inside. A path that starts
43
+ // right after a quote runs to the same quote, or to the end of the text, spaces and line breaks
44
+ // included (spec 052 AC-1 and A-052-1, spec 049 approval F-1 a third time): Node quotes the paths it
45
+ // prints, a profile folder can have a space in it, and a run error can be a stack of several lines.
46
+ const WEB_URL = /\b(?:https?|wss?|ftp):\/\/[^\s"'`<>]+/gi;
47
+ const PATH_START = /^(?:file:\/\/|[A-Za-z]:[\\/]|~?\/)/;
48
+ // A quoted path closes at the next same quote that is not inside a name (followed by a word
49
+ // character, as in O'Brien, A-052-2), across line breaks, unless that quote itself opens a
50
+ // path: then this one ends at its own line break (A-052-1, approval F-1: an unclosed quote in one
51
+ // stack line must not pair with the quote that opens the next line's path). With no such quote, the
52
+ // path runs to the end of the text.
53
+ function redactQuoted(s) {
54
+ let out = '';
55
+ let i = 0;
56
+ while (i < s.length) {
57
+ const q = s[i];
58
+ if ((q === "'" || q === '"' || q === '`') && PATH_START.test(s.slice(i + 1))) {
59
+ // A same quote followed by a word character is inside a name (O'Brien), not a closing quote,
60
+ // unless it opens a path of its own (A-052-2).
61
+ let k = s.indexOf(q, i + 1);
62
+ while (k >= 0 && /\w/.test(s[k + 1] || '') && !PATH_START.test(s.slice(k + 1))) k = s.indexOf(q, k + 1);
63
+ if (k < 0) { out += `${q}<local path>`; i = s.length; continue; }
64
+ if (!PATH_START.test(s.slice(k + 1))) { out += `${q}<local path>${q}`; i = k + 1; continue; }
65
+ const nl = s.indexOf('\n', i + 1);
66
+ const end = nl >= 0 && nl < k ? nl : k;
67
+ out += `${q}<local path>`; i = end; continue;
68
+ }
69
+ out += q; i++;
70
+ }
71
+ return out;
72
+ }
73
+ const LOCAL_PATHS = [
74
+ /\bfile:\/\/(?:[^\s"'`<>]|'(?=\w))*/gi,
75
+ /(?<![\w])[A-Za-z]:[\\/](?:[^\s"'`<>]|'(?=\w))*/g,
76
+ /(?<![\w.~-])~?\/(?:[^\s"'`<>]|'(?=\w))*[^\s"'`<>.,;:)\]]/g,
77
+ ];
78
+ function redactPaths(s) {
79
+ const urls = [];
80
+ let out = String(s).replace(WEB_URL, (u) => { urls.push(u); return `\u0000${urls.length - 1}\u0000`; });
81
+ out = redactQuoted(out);
82
+ for (const re of LOCAL_PATHS) out = out.replace(re, '<local path>');
83
+ return out.replace(/\u0000(\d+)\u0000/g, (_, n) => urls[Number(n)]);
84
+ }
85
+ // The last word on NFR-3: a receipt that still names a home directory is not written. A home
86
+ // directory is read wherever it sits in a path, not only at its start (spec 052 AC-2, spec 049
87
+ // approval F-2): /var/home and /usr/home, /mnt/c/Users and /System/Volumes/Data/Users, /var/root;
88
+ // a profile root after a drive or a WSL drive mount in either case; and home, Users or root between
89
+ // backslashes (a \\wsl$ share, a UNC path). The receipt is read as JSON, so a backslash is doubled.
90
+ // Web URLs are set aside first, as the redaction does.
91
+ const HOME_PATHS = [
92
+ /\/(?:home|Users)\/[^/\s"]+/,
93
+ /\/root\//,
94
+ /(?:[A-Za-z]:|\/mnt\/[A-Za-z])[\\/]+(?:users|documents and settings)[\\/]/i,
95
+ /\\(?:home|users|root)\\+[^\\\s"]/i,
96
+ ];
97
+ const namesHome = (text) => { const t = String(text).replace(WEB_URL, ''); return HOME_PATHS.some((re) => re.test(t)); };
98
+ const num = (x) => typeof x === 'number' && Number.isFinite(x);
99
+
100
+ // ── finding the document ─────────────────────────────────────────────────────
101
+ // A file is the document. A directory is searched, all the way down, for the one file the
102
+ // format needs; none, or more than one, is a refusal that names what was found.
103
+ function findDocument(p, from) {
104
+ const name = DOCUMENT_NAME[from];
105
+ if (!name) refuse(`unknown import source "${from}"`);
106
+ let st;
107
+ try { st = fs.statSync(p); } catch (_e) { refuse(`no such file or directory: ${p}`); }
108
+ if (st.isFile()) return p;
109
+ const found = [];
110
+ const walk = (d) => {
111
+ for (const e of fs.readdirSync(d, { withFileTypes: true }).sort((a, b) => (a.name < b.name ? -1 : 1))) {
112
+ const f = path.join(d, e.name);
113
+ if (e.isDirectory() && !e.isSymbolicLink()) walk(f);
114
+ else if (e.isFile() && e.name === name) found.push(f);
115
+ }
116
+ };
117
+ walk(p);
118
+ if (found.length === 1) return found[0];
119
+ if (!found.length) refuse(`no ${name} under ${p}: --from ${from} imports one ${name}`);
120
+ refuse(`${found.length} files named ${name} under ${p}; import one at a time:\n ${found.map((f) => path.relative(p, f)).join('\n ')}`);
121
+ }
122
+
123
+ // ── shared pieces ────────────────────────────────────────────────────────────
124
+ // One case row from its measured draws. `threshold` is the source's pass threshold, or null.
125
+ function measuredRow(id, mode, scores, threshold, judgeModel) {
126
+ const m = round(mean(scores));
127
+ const sd = round(stddev(scores));
128
+ return { id, mode, outcome: outcomeFor(m, sd, threshold), score: m, mean: m, stddev: sd, samples: scores.map((s) => round(s)), threshold, judge: { model_id: judgeModel, rubric_hash: null } };
129
+ }
130
+ function unobservedRow(id, mode) { return { id, mode, case_status: 'no_observations' }; }
131
+
132
+ function aggregateMode(rows) {
133
+ const band = aggregateBands(rows.map((c) => ({ mean: c.mean, stddev: c.stddev || 0, n: c.samples.length })));
134
+ return {
135
+ case_count: rows.length,
136
+ pass_count: rows.filter((c) => c.outcome === 'pass').length,
137
+ borderline_count: rows.filter((c) => c.outcome === 'borderline').length,
138
+ mean_score: band.mean,
139
+ stddev: band.stddev,
140
+ };
141
+ }
142
+
143
+ // Economics per arm, from what the source reports per run. `costs` are the source's own
144
+ // estimates (Format A) or absent (Format B); `wallMs` and `tokens` as reported.
145
+ function armEconomics({ n, costs = [], wallMs = [], tokens = null }) {
146
+ const q = quartiles(wallMs);
147
+ const out = {
148
+ call_count: n,
149
+ mean_input_tokens: null,
150
+ mean_output_tokens: null,
151
+ mean_cost_usd_per_call: costs.length ? round(mean(costs)) : null,
152
+ median_wall_ms: median(wallMs),
153
+ wall_ms_p25: q.p25,
154
+ wall_ms_p75: q.p75,
155
+ wall_ms_iqr: q.iqr,
156
+ };
157
+ if (tokens) out.mean_total_tokens = tokens.length ? round(mean(tokens), 2) : null;
158
+ return out;
159
+ }
160
+ function economicsBlock(basis, w, b) {
161
+ const inc = w.mean_cost_usd_per_call != null && b.mean_cost_usd_per_call != null ? round(w.mean_cost_usd_per_call - b.mean_cost_usd_per_call) : null;
162
+ return {
163
+ basis,
164
+ with_skill: w,
165
+ baseline: b,
166
+ skill_incremental_cost_usd_per_call: inc,
167
+ skill_incremental_cost_usd_per_1k_calls: inc == null ? null : round(inc * 1000),
168
+ output_tokens_delta: null,
169
+ median_wall_ms_delta: w.median_wall_ms != null && b.median_wall_ms != null ? round(w.median_wall_ms - b.median_wall_ms, 2) : null,
170
+ judge_excluded: true,
171
+ };
172
+ }
173
+
174
+ // The receipt shell. `rows` are results.cases in order; `perCase` the counts each row's
175
+ // source establishes; `excludedReasons` the reason per excluded row index.
176
+ function assemble({ tool, format, formatVersion, sourceBytes, sidecars, importedAt, notices, skill, suite, modelId, judgeModel, dateUtc, harness, rows, perCase, excludedReasons, economics }) {
177
+ const measured = rows.filter((c) => !c.case_status);
178
+ const aggW = aggregateMode(measured.filter((c) => c.mode === 'with_skill'));
179
+ const aggB = aggregateMode(measured.filter((c) => c.mode === 'baseline'));
180
+ const excluded = rows.map((c, i) => [c, i]).filter(([c]) => c.case_status).map(([c, i]) => ({ id: c.id, modes: [c.mode], reason: excludedReasons[i] }));
181
+ const imp = { tool, format, format_version: formatVersion, source_sha256: sha256(sourceBytes), imported_at: importedAt };
182
+ if (sidecars.length) imp.sidecars = sidecars;
183
+ if (notices.length) imp.notices = notices;
184
+ const receipt = {
185
+ schema_version: RECEIPT_SCHEMA_VERSION,
186
+ skill: { name: skill.name, version: skill.version, content_hash: null, unit: skill.unit },
187
+ suite: { format: suite.format, suite_hash: null, case_count: suite.caseCount },
188
+ run: {
189
+ model_id: modelId,
190
+ model_release_date: null,
191
+ provider: modelId === 'unknown' ? 'unknown' : inferProvider(modelId),
192
+ surface: 'external',
193
+ source: `imported/${tool}`,
194
+ runner_version: RUNNER_VERSION,
195
+ date_utc: dateUtc,
196
+ registry: registryStatus(modelId),
197
+ transcripts: 'none',
198
+ judge: { temperature: null, sampling: 'external', surface: 'external', model_id: judgeModel, prompt_template_hash: null },
199
+ answered_by: { kind: 'external', attested: false, reported_model: null, reported_models: null, isolation: 'none' },
200
+ ...(harness ? { harness } : {}),
201
+ import: imp,
202
+ },
203
+ results: {
204
+ cases: rows,
205
+ aggregates: { with_skill: aggW, baseline: aggB, band_rule: BAND_RULE, ...(excluded.length ? { excluded_cases: excluded } : {}) },
206
+ },
207
+ comparison: comparisonOf(aggW, aggB),
208
+ verification_level: 'DECLARED',
209
+ economics,
210
+ receipt_hash: '',
211
+ };
212
+ placeCounts(receipt, perCase);
213
+ if (namesHome(JSON.stringify(receipt))) refuse('the receipt would carry a home-directory path copied from the source; nothing is written (spec 049 NFR-3)');
214
+ return sealReceipt(receipt);
215
+ }
216
+
217
+ // ── Format A: claude plugin eval ─────────────────────────────────────────────
218
+ // The results directory's name, `2026-09-10T17-02-11-482Z`, is a UTC time.
219
+ function dateFromDirName(file) {
220
+ const m = /^(\d{4}-\d{2}-\d{2})T(\d{2})-(\d{2})-(\d{2})(?:-(\d{3}))?Z$/.exec(path.basename(path.dirname(file)));
221
+ return m ? `${m[1]}T${m[2]}:${m[3]}:${m[4]}${m[5] ? `.${m[5]}` : ''}Z` : null;
222
+ }
223
+
224
+ function importPluginEval(file, bytes, { importedAt, countErroredRuns = false }) {
225
+ let d;
226
+ try { d = JSON.parse(bytes.toString('utf8')); } catch (e) { refuse(`not JSON: ${e.message}`); }
227
+ if (!d || typeof d !== 'object' || Array.isArray(d)) refuse('expected a claude plugin eval result object');
228
+ if (d.schemaVersion !== 1) refuse(`claude plugin eval schemaVersion ${JSON.stringify(d.schemaVersion)} is not supported (this importer reads schemaVersion 1)`);
229
+ if (d.partial === true) refuse(`the document is partial, and a partial run is not imported: ${d.partialReason == null ? 'the source gives no partialReason' : String(d.partialReason)}`);
230
+ if (!Array.isArray(d.cases)) refuse('the document carries no cases[]');
231
+ const names = d.cases.map((c) => (c && typeof c.name === 'string' ? c.name : null));
232
+ if (names.some((n) => n === null)) refuse('a case carries no name');
233
+ const dup = names.find((n, i) => names.indexOf(n) !== i);
234
+ if (dup !== undefined) refuse(`two cases share the name ${JSON.stringify(dup)}; cases are never merged`);
235
+
236
+ const suite = d.suite || {};
237
+ const notices = [];
238
+ const modelId = typeof suite.modelOverride === 'string' && suite.modelOverride ? suite.modelOverride : 'unknown';
239
+ if (modelId === 'unknown') notices.push('the source records no model under test (suite.modelOverride is absent): run.model_id is unknown, never Claude Code\'s default');
240
+ const judgeModel = typeof suite.judgeModel === 'string' && suite.judgeModel ? suite.judgeModel : 'unknown';
241
+ if (judgeModel === 'unknown') notices.push('the source records no judge model (suite.judgeModel is absent); its default judge is a small fast model, so run.judge.model_id is unknown');
242
+ let dateUtc = typeof d.startedAt === 'string' ? d.startedAt : null;
243
+ if (!dateUtc) {
244
+ dateUtc = dateFromDirName(file);
245
+ notices.push(dateUtc ? 'run.date_utc is read from the results directory\'s name (the document carries no startedAt)' : 'the source carries no run date (no startedAt, and the directory name is not a results timestamp): run.date_utc is null');
246
+ }
247
+ const plugins = Array.isArray(suite.plugins) ? suite.plugins : [];
248
+ const plugin = plugins.length === 1 ? plugins[0] : null;
249
+ if (!plugin) notices.push(`the suite loads ${plugins.length} plugins, not one: skill.name and skill.version are unknown`);
250
+ const harness = { name: 'claude-code', version: typeof d.claudeVersion === 'string' ? d.claudeVersion : null };
251
+ if (harness.version === null) notices.push('the source records no claudeVersion: run.harness.version is null');
252
+ const threshold = num(suite.threshold) ? suite.threshold : null;
253
+ const twoArm = suite.ablation !== 'none';
254
+ let erroredCounted = 0;
255
+
256
+ const rows = [];
257
+ const perCase = [];
258
+ const excludedReasons = {};
259
+ const econ = { with_skill: { n: 0, costs: [], wallMs: [] }, baseline: { n: 0, costs: [], wallMs: [] } };
260
+ for (const c of d.cases) {
261
+ const arms = c.arms && typeof c.arms === 'object' ? c.arms : null;
262
+ const hasRuns = !!arms && (Array.isArray(arms.with) || Array.isArray(arms.without));
263
+ const skillGraders = (Array.isArray(c.graders) ? c.graders : []).filter((g) => g && g.type === 'tool_used' && g.config && g.config.tool === 'Skill').map((g) => g.name);
264
+ for (const [key, mode] of [['with', 'with_skill'], ['without', 'baseline']]) {
265
+ if (mode === 'baseline' && !(hasRuns ? Array.isArray(arms.without) : twoArm)) continue;
266
+ const idx = rows.length;
267
+ if (!hasRuns) {
268
+ rows.push(unobservedRow(c.name, mode));
269
+ excludedReasons[idx] = AGGREGATES_ONLY;
270
+ continue;
271
+ }
272
+ const runs = Array.isArray(arms[key]) ? arms[key] : [];
273
+ const measured = [];
274
+ const excludedDraws = [];
275
+ runs.forEach((r, i) => {
276
+ const at = i + 1;
277
+ if (r && r.error != null && !countErroredRuns) { excludedDraws.push({ draw_index: at, reason: `the source recorded an error for this run: ${redactPaths(r.error).slice(0, 300)}` }); return; }
278
+ if (r && r.skippedPaidGraders === true) { excludedDraws.push({ draw_index: at, reason: 'the source skipped this run\'s paid graders (skippedPaidGraders: true), so its score omits them' }); return; }
279
+ if (!r || !num(r.score)) { excludedDraws.push({ draw_index: at, reason: 'the source gives this run no numeric score' }); return; }
280
+ if (r.error != null) erroredCounted++;
281
+ measured.push(r);
282
+ });
283
+ perCase.push({ index: idx, counts: { generations_per_arm: measured.length } });
284
+ if (!measured.length) {
285
+ const row = unobservedRow(c.name, mode);
286
+ if (excludedDraws.length) row.excluded_draws = excludedDraws;
287
+ rows.push(row);
288
+ excludedReasons[idx] = runs.length ? 'every run of this arm was kept out; each is listed in excluded_draws' : 'the source lists no runs for this arm';
289
+ continue;
290
+ }
291
+ const row = measuredRow(c.name, mode, measured.map((r) => r.score), threshold, judgeModel);
292
+ if (excludedDraws.length) row.excluded_draws = excludedDraws;
293
+ // Activation: an unscored tool_used Skill grader says whether the skill fired. It is
294
+ // recorded beside the score and never enters it (the run's score already excludes it).
295
+ if (mode === 'with_skill') {
296
+ const activation = [];
297
+ for (const name of skillGraders) {
298
+ const seen = measured.map((r) => (Array.isArray(r.graders) ? r.graders.find((g) => g && g.name === name) : null)).filter((g) => g && g.scored === false);
299
+ if (seen.length) activation.push({ indicator: name, fired: seen.filter((g) => g.passed === true).length, runs: seen.length });
300
+ }
301
+ if (activation.length) row.activation = activation;
302
+ }
303
+ rows.push(row);
304
+ econ[mode].n += measured.length;
305
+ for (const r of measured) {
306
+ if (num(r.costUsd)) econ[mode].costs.push(r.costUsd);
307
+ if (num(r.durationSeconds)) econ[mode].wallMs.push(Math.round(r.durationSeconds * 1000));
308
+ }
309
+ }
310
+ }
311
+ if (erroredCounted) notices.push(`${erroredCounted} errored run(s) are counted as measured draws (--count-errored-runs)`);
312
+ return assemble({
313
+ tool: 'claude-plugin-eval', format: 'aggregate-result.json', formatVersion: d.schemaVersion, sourceBytes: bytes, sidecars: [], importedAt, notices,
314
+ skill: { name: plugin && typeof plugin.name === 'string' ? plugin.name : 'unknown', version: plugin && typeof plugin.version === 'string' ? plugin.version : 'unknown', unit: 'plugin' },
315
+ suite: { format: 'claude-plugin-eval/aggregate-result.json', caseCount: d.cases.length },
316
+ modelId, judgeModel, dateUtc, harness, rows, perCase, excludedReasons,
317
+ economics: economicsBlock('source-list-price-estimate', armEconomics(econ.with_skill), armEconomics(econ.baseline)),
318
+ });
319
+ }
320
+
321
+ // ── Format B: benchmark.json (skill-creator, skill-up) ───────────────────────
322
+ const CONFIGURATIONS = { with_skill: 'with_skill', without_skill: 'baseline' };
323
+
324
+ // skill-up's result.json names the model it forwarded (applied_configuration.model) and,
325
+ // only when the agent reported one, the model observed. Observed first.
326
+ function modelFromResult(res) {
327
+ const oc = res && res.observed_configuration;
328
+ if (oc && typeof oc.model === 'string' && oc.model) return { model: oc.model, from: 'observed_configuration.model' };
329
+ const per = (Array.isArray(res && res.case_results) ? res.case_results : []).map((c) => c && c.observed_model);
330
+ if (per.length && per.every((m) => typeof m === 'string' && m && m === per[0])) return { model: per[0], from: 'case_results[].observed_model' };
331
+ const ac = res && res.applied_configuration;
332
+ if (ac && typeof ac.model === 'string' && ac.model) return { model: ac.model, from: 'applied_configuration.model' };
333
+ return null;
334
+ }
335
+
336
+ function importBenchmark(file, bytes, { importedAt }) {
337
+ let d;
338
+ try { d = JSON.parse(bytes.toString('utf8')); } catch (e) { refuse(`not JSON: ${e.message}`); }
339
+ if (!d || typeof d !== 'object' || !d.metadata || typeof d.metadata !== 'object') refuse('expected a benchmark.json with metadata');
340
+ const meta = d.metadata;
341
+ const runs = Array.isArray(d.runs) ? d.runs : [];
342
+ const notices = [];
343
+ const groups = new Map();
344
+ const labels = new Map();
345
+ const seen = new Set();
346
+ for (const r of runs) {
347
+ if (!r || (typeof r.eval_id !== 'number' && typeof r.eval_id !== 'string')) refuse('a runs[] row carries no eval_id');
348
+ const mode = CONFIGURATIONS[r.configuration];
349
+ if (!mode) refuse(`configuration ${JSON.stringify(r.configuration)} is neither with_skill nor without_skill`);
350
+ const id = String(r.eval_id);
351
+ const key = `${id}\u0000${r.configuration}\u0000${r.run_number}`;
352
+ if (seen.has(key)) refuse(`two rows share eval_id ${id}, configuration ${r.configuration} and run_number ${r.run_number}; rows are never merged`);
353
+ seen.add(key);
354
+ if (!groups.has(id)) groups.set(id, { with_skill: [], baseline: [] });
355
+ groups.get(id)[mode].push(r);
356
+ if (typeof r.eval_name === 'string' && !labels.has(id)) labels.set(id, r.eval_name);
357
+ }
358
+ // Evals the metadata names with no row are aggregates only.
359
+ const declared = (Array.isArray(meta.evals_run) ? meta.evals_run : []).map(String);
360
+ const ids = [...new Set([...groups.keys(), ...declared])];
361
+ const hasBaseline = runs.some((r) => r.configuration === 'without_skill') || !!(d.run_summary && d.run_summary.without_skill);
362
+
363
+ // The model, from the document or the result.json beside it.
364
+ const sidecars = [];
365
+ let modelId = typeof meta.executor_model === 'string' && meta.executor_model ? meta.executor_model : null;
366
+ let harness = null;
367
+ const resultFile = path.join(path.dirname(file), 'result.json');
368
+ let res = null;
369
+ if (fs.existsSync(resultFile)) {
370
+ const rb = fs.readFileSync(resultFile);
371
+ try { res = JSON.parse(rb.toString('utf8')); sidecars.push({ file: 'result.json', sha256: sha256(rb) }); } catch (_e) { notices.push('result.json beside the document does not parse and is not read'); }
372
+ }
373
+ if (!modelId) {
374
+ const found = res && modelFromResult(res);
375
+ if (found) {
376
+ modelId = found.model;
377
+ notices.push(found.from === 'applied_configuration.model'
378
+ ? 'run.model_id is the model skill-up forwarded to the agent (result.json applied_configuration.model); the agent reported none, so it is not an observation'
379
+ : `run.model_id is read from result.json ${found.from}`);
380
+ } else {
381
+ modelId = 'unknown';
382
+ notices.push(res ? 'neither benchmark.json nor result.json records the model: run.model_id is unknown' : 'benchmark.json records no executor_model and no result.json is beside it: run.model_id is unknown');
383
+ }
384
+ }
385
+ if (res && typeof res.engine_name === 'string' && res.engine_name) {
386
+ const v = res.observed_configuration && typeof res.observed_configuration.version === 'string' ? res.observed_configuration.version : null;
387
+ harness = { name: res.engine_name, version: v };
388
+ }
389
+ // No cited benchmark.json format records the grader's model (skill-up's grading.json carries
390
+ // expectations and a summary only), and analyzer_model is not the grader, so the judge is unknown.
391
+ const judgeModel = 'unknown';
392
+ notices.push('benchmark.json records no grader model: run.judge.model_id is unknown (analyzer_model is not the grader and is not read)');
393
+ const dateUtc = typeof meta.timestamp === 'string' && meta.timestamp ? meta.timestamp : null;
394
+ if (!dateUtc) notices.push('benchmark.json carries no metadata.timestamp: run.date_utc is null');
395
+
396
+ const rows = [];
397
+ const perCase = [];
398
+ const excludedReasons = {};
399
+ const econ = { with_skill: { n: 0, wallMs: [], tokens: [] }, baseline: { n: 0, wallMs: [], tokens: [] } };
400
+ const observed = new Set();
401
+ for (const id of ids) {
402
+ const g = groups.get(id);
403
+ for (const mode of ['with_skill', 'baseline']) {
404
+ if (mode === 'baseline' && !hasBaseline) continue;
405
+ const idx = rows.length;
406
+ const rs = g ? g[mode].slice().sort((a, b) => Number(a.run_number) - Number(b.run_number)) : [];
407
+ const scored = rs.filter((r) => r.result && num(r.result.pass_rate));
408
+ if (!g) {
409
+ rows.push(unobservedRow(id, mode));
410
+ excludedReasons[idx] = AGGREGATES_ONLY;
411
+ continue;
412
+ }
413
+ perCase.push({ index: idx, counts: { generations_per_arm: scored.length } });
414
+ observed.add(scored.length);
415
+ const row = scored.length ? measuredRow(id, mode, scored.map((r) => r.result.pass_rate), null, judgeModel) : unobservedRow(id, mode);
416
+ if (labels.has(id)) row.label = labels.get(id);
417
+ const dropped = rs.filter((r) => !scored.includes(r)).map((r) => ({ draw_index: Number(r.run_number), reason: 'the source gives this run no numeric pass_rate' }));
418
+ if (dropped.length) row.excluded_draws = dropped;
419
+ rows.push(row);
420
+ if (!scored.length) { excludedReasons[idx] = 'the source lists no scored run for this arm'; continue; }
421
+ econ[mode].n += scored.length;
422
+ for (const r of scored) {
423
+ if (num(r.result.time_seconds)) econ[mode].wallMs.push(Math.round(r.result.time_seconds * 1000));
424
+ if (num(r.result.tokens)) econ[mode].tokens.push(r.result.tokens);
425
+ }
426
+ }
427
+ }
428
+ if (num(meta.runs_per_configuration) && [...observed].some((n) => n !== meta.runs_per_configuration)) {
429
+ notices.push(`metadata.runs_per_configuration declares ${meta.runs_per_configuration}; the rows carry ${[...observed].sort().join(', ')} per configuration, and the rows are what is recorded`);
430
+ }
431
+ return assemble({
432
+ tool: 'skill-creator', format: 'benchmark.json', formatVersion: null, sourceBytes: bytes, sidecars, importedAt, notices,
433
+ skill: { name: typeof meta.skill_name === 'string' && meta.skill_name ? meta.skill_name : 'unknown', version: 'unknown', unit: 'skill' },
434
+ suite: { format: 'skill-creator/benchmark.json', caseCount: ids.length },
435
+ modelId, judgeModel, dateUtc, harness, rows, perCase, excludedReasons,
436
+ economics: economicsBlock('source-reported', armEconomics(econ.with_skill), armEconomics(econ.baseline)),
437
+ });
438
+ }
439
+
440
+ // Import one document. `p` is the file or directory the user named.
441
+ function importAnthropic(p, { from, importedAt, countErroredRuns = false } = {}) {
442
+ if (!ANTHROPIC_TOOLS.includes(from)) refuse(`unknown import source "${from}"`);
443
+ const file = findDocument(p, from);
444
+ const bytes = fs.readFileSync(file);
445
+ const receipt = from === 'claude-plugin-eval'
446
+ ? importPluginEval(file, bytes, { importedAt, countErroredRuns })
447
+ : importBenchmark(file, bytes, { importedAt });
448
+ return { receipt, file };
449
+ }
450
+
451
+ module.exports = { importAnthropic, findDocument, ANTHROPIC_TOOLS, DOCUMENT_NAME, ImportRefused };
package/lib/importers.js CHANGED
@@ -22,6 +22,7 @@ const { mean, stddev, round, combineUncertainty, aggregateBands } = require('./s
22
22
  const { outcomeFor } = require('./run');
23
23
  const { inferProvider } = require('./provider');
24
24
  const { registryStatus } = require('./models');
25
+ const { placeCounts } = require('./counts');
25
26
 
26
27
  const IMPORT_TOOLS = ['agent-skills-eval', 'skillgrade'];
27
28
 
@@ -41,10 +42,17 @@ function aggregateMode(cases) {
41
42
  }
42
43
 
43
44
  // Shared receipt shell for both importers. `judgeBlock` describes the SOURCE
44
- // tool's grading (samples = what it actually did), never our sampled judge.
45
- function importedReceipt({ tool, skillName, skillVersion, suiteFormat, caseCount, modelId, dateUtc, judgeBlock, cases, comparison }) {
46
- const withSkill = cases.filter((c) => c.mode === 'with_skill');
47
- const baseline = cases.filter((c) => c.mode === 'baseline');
45
+ // tool's grading, never our sampled judge. v0.8 (spec 043): it carries no sample
46
+ // count, because neither source format defines one; `perCase` carries only the counts
47
+ // the source establishes, and placeCounts writes them at the narrowest honest scope.
48
+ // A case with no observation (case_status no_observations) is listed, excluded from
49
+ // every aggregate, and named in excluded_cases with `excludedReasons[id]`.
50
+ function importedReceipt({ tool, skillName, skillVersion, suiteFormat, caseCount, modelId, dateUtc, judgeBlock, cases, comparison, perCase = [], excludedReasons = {} }) {
51
+ const measured = cases.filter((c) => !c.case_status || c.case_status === 'ok');
52
+ const withSkill = measured.filter((c) => c.mode === 'with_skill');
53
+ const baseline = measured.filter((c) => c.mode === 'baseline');
54
+ const excluded = cases.filter((c) => c.case_status && c.case_status !== 'ok')
55
+ .map((c) => ({ id: c.id, modes: [c.mode], reason: excludedReasons[c.id] || 'the source supplied no observation for this case' }));
48
56
  const receipt = {
49
57
  schema_version: RECEIPT_SCHEMA_VERSION,
50
58
  skill: {
@@ -73,12 +81,13 @@ function importedReceipt({ tool, skillName, skillVersion, suiteFormat, caseCount
73
81
  },
74
82
  results: {
75
83
  cases,
76
- aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline), band_rule: BAND_RULE },
84
+ aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline), band_rule: BAND_RULE, ...(excluded.length ? { excluded_cases: excluded } : {}) },
77
85
  },
78
86
  comparison,
79
87
  verification_level: 'DECLARED',
80
88
  receipt_hash: '',
81
89
  };
90
+ placeCounts(receipt, perCase);
82
91
  return sealReceipt(receipt);
83
92
  }
84
93
 
@@ -139,7 +148,9 @@ function importAgentSkillsEval(data, { importedAt } = {}) {
139
148
  caseCount: data.evals.length,
140
149
  modelId: data.target || 'unknown',
141
150
  dateUtc: data.timestamp || importedAt || new Date().toISOString(),
142
- judgeBlock: { samples: 1, temperature: null, sampling: 'external', surface: 'external' },
151
+ // No count of either kind: the agent-skills-eval benchmark format defines neither
152
+ // (spec 043 AC-5, A-1), so none is written until it does.
153
+ judgeBlock: { temperature: null, sampling: 'external', surface: 'external' },
143
154
  cases,
144
155
  comparison,
145
156
  });
@@ -158,16 +169,38 @@ function importSkillgrade(data, { importedAt } = {}) {
158
169
  const graderModel = data.grader_model || 'unknown';
159
170
  const defaultThreshold = typeof data.threshold === 'number' ? data.threshold : 0.8;
160
171
  const cases = [];
161
- let maxTrials = 1;
172
+ const perCase = [];
173
+ const excludedReasons = {};
162
174
  for (const task of data.tasks) {
163
- const rewards = (task.trials || []).map((t) => (typeof t === 'number' ? t : t.reward)).filter((r) => typeof r === 'number');
164
- if (!rewards.length) continue;
165
- maxTrials = Math.max(maxTrials, rewards.length);
175
+ const id = String(task.name);
176
+ // Spec 043 A-3: keep the distinction the source makes. A trials list that is
177
+ // present and empty is zero observations; no trials field is an unknown count.
178
+ // Either way the task is listed and contributes no figure.
179
+ if (!Array.isArray(task.trials)) {
180
+ cases.push({ id, mode: 'with_skill', case_status: 'no_observations' });
181
+ excludedReasons[id] = 'the source carries no trials field for this task: its trial count is unknown';
182
+ continue;
183
+ }
184
+ const rewards = task.trials.map((t) => (typeof t === 'number' ? t : t.reward)).filter((r) => typeof r === 'number');
185
+ if (!task.trials.length) {
186
+ perCase.push({ index: cases.length, counts: { generations_per_arm: 0 } });
187
+ cases.push({ id, mode: 'with_skill', case_status: 'no_observations', samples: [] });
188
+ excludedReasons[id] = 'the source lists no trials for this task (trials is empty): zero observations';
189
+ continue;
190
+ }
191
+ if (!rewards.length) {
192
+ cases.push({ id, mode: 'with_skill', case_status: 'no_observations' });
193
+ excludedReasons[id] = 'the source lists trials for this task but none carries a numeric reward';
194
+ continue;
195
+ }
196
+ // Trials are independent generations (A-2): their number is this case's
197
+ // generation count, and never a judge-sample count.
198
+ perCase.push({ index: cases.length, counts: { generations_per_arm: rewards.length } });
166
199
  const m = round(mean(rewards));
167
200
  const sd = round(stddev(rewards));
168
201
  const threshold = typeof task.threshold === 'number' ? task.threshold : defaultThreshold;
169
202
  cases.push({
170
- id: String(task.name),
203
+ id,
171
204
  mode: 'with_skill',
172
205
  // Same outcome rule as a Driftproof run (borderline when the threshold
173
206
  // sits inside mean ± stddev) — a deterministic READING of their numbers.
@@ -188,11 +221,15 @@ function importSkillgrade(data, { importedAt } = {}) {
188
221
  // imported verbatim (or the results' model field when present).
189
222
  modelId: data.model || data.agent || 'unknown',
190
223
  dateUtc: data.timestamp || importedAt || new Date().toISOString(),
191
- judgeBlock: { samples: maxTrials, temperature: null, sampling: 'external', surface: 'external' },
224
+ // No judge-sample count: the skillgrade format defines none, and a trial count is a
225
+ // generation count (spec 043 AC-5, A-1, A-2).
226
+ judgeBlock: { temperature: null, sampling: 'external', surface: 'external' },
192
227
  cases,
228
+ perCase,
229
+ excludedReasons,
193
230
  // No baseline mode exists in skillgrade — nulls, never a fabricated 0.
194
231
  comparison: {
195
- with_skill_score: round(mean(cases.map((c) => c.mean))),
232
+ with_skill_score: round(mean(cases.filter((c) => !c.case_status).map((c) => c.mean))),
196
233
  baseline_score: null,
197
234
  delta: null,
198
235
  delta_uncertainty: null,
package/lib/init.js CHANGED
@@ -11,7 +11,9 @@ const path = require('path');
11
11
  // <dir>/evals/evals.json — an eval suite with 3 example cases, each rubric
12
12
  // anchored at 0.80 (the scoring convention the
13
13
  // example suite and Report #001 use)
14
- // <dir>/.driftproofrc — per-project run defaults (budget, models, samples)
14
+ // <dir>/.driftproofrc - per-project run defaults (budget, models). No samples,
15
+ // max_cases or judge_model: a skill directory's rc may not
16
+ // set them, and `run` ignores them there (spec 062).
15
17
  //
16
18
  // NEVER overwrites an existing file — every write is guarded, and an existing
17
19
  // path is reported as "skipped". Safe to re-run.
@@ -89,7 +91,6 @@ function rcTemplate() {
89
91
  return {
90
92
  _comment: 'Per-project driftproof defaults. CLI flags override these. See `driftproof help`.',
91
93
  models: 'claude-haiku-4-5',
92
- samples: 5,
93
94
  max_usd: 2,
94
95
  };
95
96
  }