@uinaf/skillcheck 1.0.1 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -28,7 +28,8 @@ pre-npm git tags are covered there too.
28
28
  ```sh
29
29
  skillcheck lint # structural lint, no credentials
30
30
  skillcheck run skills/wat/evals/basic # one scenario, graded end to end
31
- skillcheck sweep && skillcheck summarize # every scenario, then a scorecard
31
+ skillcheck sweep --trials 3 # every scenario, three trials each
32
+ skillcheck summarize # per-scenario and per-skill scorecard
32
33
  ```
33
34
 
34
35
  `lint` needs nothing. `run` and `sweep` need model auth, which is why sweeps
package/dist/cli.js CHANGED
@@ -1,6 +1,6 @@
1
1
  #!/usr/bin/env node
2
2
  import { lintSkills } from "./lint.js";
3
- import { generateRun, requiredEvalPackages, resolvePackageDir, runNameFor } from "./scenario.js";
3
+ import { encodeRunNamePart, generateRun, requiredEvalPackages, resolvePackageDir, runNameFor } from "./scenario.js";
4
4
  import { execFileSync, spawnSync } from "node:child_process";
5
5
  import fs from "node:fs";
6
6
  import path from "node:path";
@@ -12,11 +12,18 @@ const selfExt = path.extname(fileURLToPath(import.meta.url));
12
12
  function toolVersion() {
13
13
  return JSON.parse(fs.readFileSync(path.join(packageDir, "package.json"), "utf8")).version;
14
14
  }
15
- function parseMaxTurns(raw) {
15
+ function parsePositiveInt(flag, raw) {
16
16
  const n = Number(raw);
17
- if (!Number.isInteger(n) || n <= 0) throw new Error(`--max-turns must be a positive integer, got ${JSON.stringify(raw)}`);
17
+ if (!Number.isInteger(n) || n <= 0) throw new Error(`${flag} must be a positive integer, got ${JSON.stringify(raw)}`);
18
18
  return n;
19
19
  }
20
+ const AGENT_EFFORTS = [
21
+ "low",
22
+ "medium",
23
+ "high",
24
+ "xhigh",
25
+ "max"
26
+ ];
20
27
  function fail(msg) {
21
28
  console.error(msg);
22
29
  process.exit(1);
@@ -29,8 +36,10 @@ function parseArgs(argv) {
29
36
  "--agent",
30
37
  "--judge",
31
38
  "--judge-effort",
39
+ "--agent-effort",
32
40
  "--harness",
33
- "--max-turns"
41
+ "--max-turns",
42
+ "--trials"
34
43
  ]);
35
44
  for (let i = 0; i < argv.length; i++) {
36
45
  const a = argv[i];
@@ -60,8 +69,13 @@ function stateDirs(root) {
60
69
  }
61
70
  function runOptions(flags) {
62
71
  const harness = flags.get("--harness") ?? "claude";
63
- if (harness !== "claude" && harness !== "codex" && harness !== "grok") fail(`--harness must be claude, codex, or grok, got ${harness}`);
64
- if (flags.has("--max-turns") && harness !== "claude") fail("--max-turns is only supported with --harness claude");
72
+ if (harness !== "claude" && harness !== "codex" && harness !== "grok") throw new Error(`--harness must be claude, codex, or grok, got ${harness}`);
73
+ if (flags.has("--max-turns") && harness !== "claude") throw new Error("--max-turns is only supported with --harness claude");
74
+ const agentEffort = flags.get("--agent-effort");
75
+ if (agentEffort !== void 0) {
76
+ if (harness !== "claude") throw new Error(`--agent-effort is only supported with --harness claude; ${harness} has no effort setting wired`);
77
+ if (!AGENT_EFFORTS.includes(agentEffort)) throw new Error(`--agent-effort must be ${AGENT_EFFORTS.join(", ")}, got ${agentEffort}`);
78
+ }
65
79
  const agent = flags.get("--agent");
66
80
  const judgeModel = flags.get("--judge") ?? "claude-opus-5";
67
81
  const judgeEffort = flags.get("--judge-effort");
@@ -71,17 +85,51 @@ function runOptions(flags) {
71
85
  "low",
72
86
  "medium",
73
87
  "high"
74
- ].includes(judgeEffort)) fail(`--judge-effort must be minimal, low, medium, or high, got ${judgeEffort}`);
75
- if (!judgeModel.includes(":")) fail("--judge-effort needs a provider-qualified --judge (e.g. openai:chat:gpt-5.6-sol); the Anthropic judge does not take a reasoning effort");
88
+ ].includes(judgeEffort)) throw new Error(`--judge-effort must be minimal, low, medium, or high, got ${judgeEffort}`);
89
+ if (!judgeModel.includes(":")) throw new Error("--judge-effort needs a provider-qualified --judge (e.g. openai:chat:gpt-5.6-sol); the Anthropic judge does not take a reasoning effort");
76
90
  }
77
91
  return {
78
92
  harness,
79
93
  agentModel: agent,
94
+ agentEffort,
80
95
  judgeModel,
81
96
  judgeEffort,
82
- maxTurns: flags.has("--max-turns") ? parseMaxTurns(flags.get("--max-turns")) : void 0
97
+ maxTurns: flags.has("--max-turns") ? parsePositiveInt("--max-turns", flags.get("--max-turns")) : void 0,
98
+ trials: flags.has("--trials") ? parsePositiveInt("--trials", flags.get("--trials")) : 1
99
+ };
100
+ }
101
+ function runConfigOf(opts) {
102
+ return {
103
+ agent_model: opts.agentModel ?? (opts.harness === "claude" ? "claude-opus-5" : `${opts.harness}-default`),
104
+ agent_effort: opts.agentEffort ?? null,
105
+ judge_model: opts.judgeModel,
106
+ judge_effort: opts.judgeEffort ?? null,
107
+ trials: opts.trials ?? 1
83
108
  };
84
109
  }
110
+ function configKey(c) {
111
+ return JSON.stringify([
112
+ c.agent_model,
113
+ c.agent_effort ?? null,
114
+ c.judge_model,
115
+ c.judge_effort ?? null,
116
+ c.trials ?? 1
117
+ ]);
118
+ }
119
+ function describeConfig(c) {
120
+ const effort = (e) => e ? `@${e}` : "";
121
+ return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}`;
122
+ }
123
+ function assertUniformConfig(entries, allowMixed) {
124
+ if (allowMixed) return;
125
+ const byHarness = /* @__PURE__ */ new Map();
126
+ for (const e of entries) {
127
+ const configs = byHarness.get(e.harness) ?? /* @__PURE__ */ new Map();
128
+ configs.set(configKey(e), e);
129
+ byHarness.set(e.harness, configs);
130
+ }
131
+ for (const [harness, configs] of byHarness) if (configs.size > 1) throw new Error(`${harness} results span multiple run configurations (${[...configs.values()].map(describeConfig).join("; ")}); rerun them to match or pass --allow-mixed`);
132
+ }
85
133
  function ensureEvalPackages(opts) {
86
134
  const missing = requiredEvalPackages(opts, process.env.ANTHROPIC_API_KEY !== void 0).filter((pkg) => resolvePackageDir(pkg) === void 0);
87
135
  if (missing.length === 0) return;
@@ -105,21 +153,76 @@ function promptfooEntry() {
105
153
  if (typeof rel !== "string") throw new Error(`promptfoo at ${dir} declares no bin`);
106
154
  return path.join(dir, rel);
107
155
  }
108
- function classifyResult(raw) {
156
+ const GRADED_REASONS = /* @__PURE__ */ new Set([0, 1]);
157
+ const ERRORED_REASON = 2;
158
+ function classifyResult(raw, expectedTrials) {
109
159
  const root = raw;
110
- const res = root?.results?.results?.[0];
160
+ const rows = root?.results?.results;
161
+ if (!Array.isArray(rows) || rows.length === 0) return { error: "promptfoo output carried no result" };
162
+ if (expectedTrials !== void 0 && rows.length !== expectedTrials) return { error: `promptfoo returned ${rows.length} of ${expectedTrials} trials` };
163
+ const stats = rows.length === 1 ? root?.results?.stats ?? void 0 : void 0;
164
+ const trials = [];
165
+ for (const [i, row] of rows.entries()) {
166
+ const verdict = classifyRow(row, stats);
167
+ if ("error" in verdict) return { error: rows.length === 1 ? verdict.error : `trial ${i + 1}: ${verdict.error}` };
168
+ trials.push(verdict);
169
+ }
170
+ return { trials };
171
+ }
172
+ function classifyRow(raw, stats) {
173
+ const res = raw;
111
174
  if (res === void 0 || res === null) return { error: "promptfoo output carried no result" };
112
175
  const message = typeof res.error === "string" ? res.error.trim() : "";
113
- const stats = root?.results?.stats ?? void 0;
114
- const gradedByStats = stats !== void 0 && (stats.errors ?? 0) === 0 && ((stats.failures ?? 0) > 0 || (stats.successes ?? 0) > 0);
115
- if (message !== "" && !gradedByStats) return { error: message };
116
- if (stats !== void 0 && (stats.errors ?? 0) > 0 && (stats.successes ?? 0) === 0 && (stats.failures ?? 0) === 0) return { error: "promptfoo reported an errored test with nothing graded" };
176
+ if (res.failureReason === ERRORED_REASON) return { error: message || "promptfoo reported an errored test" };
177
+ const graded = GRADED_REASONS.has(res.failureReason) || stats !== void 0 && (stats.errors ?? 0) === 0 && ((stats.failures ?? 0) > 0 || (stats.successes ?? 0) > 0);
178
+ if (message !== "" && !graded) return { error: message };
179
+ if (!graded && stats !== void 0 && (stats.errors ?? 0) > 0 && (stats.successes ?? 0) === 0 && (stats.failures ?? 0) === 0) return { error: "promptfoo reported an errored test with nothing graded" };
117
180
  if (typeof res.score !== "number" || typeof res.success !== "boolean") return { error: "promptfoo result carried no usable score" };
181
+ const components = Array.isArray(res.gradingResult?.componentResults) ? res.gradingResult.componentResults : [];
182
+ const checklist = components.find((c) => c?.metadata?.assertionSet?.type === "assert-set");
183
+ const skillUsed = components.find((c) => c?.assertion?.type === "skill-used");
184
+ if (typeof checklist?.score !== "number" || typeof skillUsed?.pass !== "boolean") return { error: "promptfoo result carried no checklist or skill-used verdict" };
118
185
  return {
119
- score: res.score,
120
- pass: res.success
186
+ score: checklist.score,
187
+ pass: res.success,
188
+ skillUsed: skillUsed.pass
121
189
  };
122
190
  }
191
+ const NOISY_SPREAD = .2;
192
+ const round4 = (n) => Math.round(n * 1e4) / 1e4;
193
+ function aggregateTrials(trials) {
194
+ if (trials.length === 0) throw new Error("cannot aggregate zero trials");
195
+ const scores = trials.map((t) => t.score);
196
+ const passes = trials.filter((t) => t.pass).length;
197
+ const min = Math.min(...scores);
198
+ const spread = Math.max(...scores) - min;
199
+ const skillUsed = trials.filter((t) => t.skillUsed).length;
200
+ return {
201
+ trials: trials.length,
202
+ pass: passes === trials.length,
203
+ passes,
204
+ pass_rate: round4(passes / trials.length),
205
+ score: round4(scores.reduce((a, b) => a + b, 0) / trials.length),
206
+ score_min: round4(min),
207
+ score_spread: round4(spread),
208
+ skill_used: skillUsed,
209
+ skill_used_rate: round4(skillUsed / trials.length),
210
+ noisy: passes > 0 && passes < trials.length || spread >= .199999999
211
+ };
212
+ }
213
+ function formatStats(s) {
214
+ const line = `score=${s.score.toFixed(4)}`;
215
+ if (s.trials === 1) return line;
216
+ return [
217
+ line,
218
+ `min=${s.score_min.toFixed(4)}`,
219
+ `spread=${s.score_spread.toFixed(4)}`,
220
+ `pass^${s.trials}=${s.pass ? "yes" : "no"}`,
221
+ `passes=${s.passes}/${s.trials}`,
222
+ `skill-used=${s.skill_used}/${s.trials}`,
223
+ ...s.noisy ? ["NOISY"] : []
224
+ ].join(" ");
225
+ }
123
226
  function gitHead(root) {
124
227
  return execFileSync("git", ["rev-parse", "HEAD"], {
125
228
  cwd: root,
@@ -134,14 +237,19 @@ function attemptPath(resultPath) {
134
237
  }
135
238
  function runScenario(scenarioDir, opts, root) {
136
239
  const dirs = stateDirs(root);
137
- const { name, configPath } = generateRun(path.resolve(scenarioDir), opts, {
240
+ const { name, configPath, skill, scenario } = generateRun(path.resolve(scenarioDir), opts, {
138
241
  scratchDir: dirs.scratch,
139
242
  transformPath: path.join(here, `transform${selfExt}`),
140
243
  grokProviderPath: path.join(here, `grok-provider${selfExt}`)
141
244
  });
142
245
  fs.mkdirSync(dirs.results, { recursive: true });
143
246
  const resultPath = path.join(dirs.results, `${name}.json`);
144
- fs.writeFileSync(attemptPath(resultPath), "{}\n");
247
+ const identity = {
248
+ skill,
249
+ scenario,
250
+ harness: opts.harness
251
+ };
252
+ fs.writeFileSync(attemptPath(resultPath), JSON.stringify(identity) + "\n");
145
253
  fs.rmSync(resultPath, { force: true });
146
254
  fs.rmSync(metaPath(resultPath), { force: true });
147
255
  const sha = gitHead(root);
@@ -172,24 +280,60 @@ function runScenario(scenarioDir, opts, root) {
172
280
  if (rc !== 0) return outcome;
173
281
  let verdict;
174
282
  try {
175
- verdict = classifyResult(JSON.parse(fs.readFileSync(resultPath, "utf8")));
283
+ verdict = classifyResult(JSON.parse(fs.readFileSync(resultPath, "utf8")), opts.trials ?? 1);
176
284
  } catch {
177
285
  verdict = { error: "promptfoo produced no parseable result file" };
178
286
  }
179
- outcome.score = verdict.score;
180
- outcome.pass = verdict.pass;
181
- outcome.error = verdict.error;
182
- if (verdict.score !== void 0) {
183
- fs.writeFileSync(metaPath(resultPath), JSON.stringify({
184
- skills_tree_sha: sha,
185
- harness: opts.harness,
186
- ran_at: (/* @__PURE__ */ new Date()).toISOString(),
187
- tool_version: toolVersion()
188
- }, null, 2) + "\n");
189
- fs.rmSync(attemptPath(resultPath), { force: true });
190
- }
287
+ if ("error" in verdict) return {
288
+ ...outcome,
289
+ error: verdict.error
290
+ };
291
+ outcome.stats = aggregateTrials(verdict.trials);
292
+ fs.writeFileSync(metaPath(resultPath), JSON.stringify({
293
+ skills_tree_sha: sha,
294
+ ...identity,
295
+ ...runConfigOf(opts),
296
+ aggregate: outcome.stats,
297
+ ran_at: (/* @__PURE__ */ new Date()).toISOString(),
298
+ tool_version: toolVersion()
299
+ }, null, 2) + "\n");
300
+ fs.rmSync(attemptPath(resultPath), { force: true });
191
301
  return outcome;
192
302
  }
303
+ function judgeName(judge) {
304
+ if (typeof judge === "string") return judge.replace(/^anthropic:messages:/, "");
305
+ const j = judge;
306
+ const name = j?.config?.model ?? j?.id;
307
+ return typeof name === "string" ? name : "unknown";
308
+ }
309
+ function resultRunConfig(raw, meta, harness) {
310
+ const m = meta;
311
+ if (typeof m?.agent_model === "string" && typeof m.judge_model === "string") return {
312
+ agent_model: m.agent_model,
313
+ agent_effort: m.agent_effort ?? null,
314
+ judge_model: m.judge_model,
315
+ judge_effort: m.judge_effort ?? null,
316
+ trials: m.trials ?? 1
317
+ };
318
+ const r = raw;
319
+ const agent = r?.config?.providers?.[0]?.config;
320
+ const judge = r?.config?.defaultTest?.options?.provider;
321
+ const judgeEffort = judge?.config?.reasoning_effort;
322
+ return {
323
+ agent_model: typeof agent?.model === "string" ? agent.model : `${harness}-default`,
324
+ agent_effort: typeof agent?.effort === "string" ? agent.effort : null,
325
+ judge_model: judgeName(judge),
326
+ judge_effort: typeof judgeEffort === "string" ? judgeEffort : null,
327
+ trials: r?.results?.results?.length ?? 1
328
+ };
329
+ }
330
+ function readJson(file) {
331
+ try {
332
+ return JSON.parse(fs.readFileSync(file, "utf8"));
333
+ } catch {
334
+ return;
335
+ }
336
+ }
193
337
  function discoverScenarios(root) {
194
338
  const roots = [path.join(root, "skills")];
195
339
  const cliDir = path.join(root, "cli");
@@ -210,61 +354,79 @@ function discoverScenarios(root) {
210
354
  }
211
355
  return found.sort();
212
356
  }
357
+ const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N]";
213
358
  function cmdRun(argv) {
214
359
  const { positional, flags } = parseArgs(argv);
215
- if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--judge-effort EFFORT] [--harness claude|codex|grok]");
360
+ if (positional.length !== 1) fail(`usage: skillcheck run <scenario-dir> ${RUN_FLAGS}`);
216
361
  const opts = runOptions(flags);
217
362
  ensureEvalPackages(opts);
218
363
  const o = runScenario(positional[0], opts, resolveRoot(flags));
219
- if (o.score === void 0) {
364
+ if (o.stats === void 0) {
220
365
  console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
221
366
  process.exit(2);
222
367
  }
223
- console.log(`${o.pass ? "PASS" : "FAIL"} ${o.name} score=${o.score.toFixed(4)} (results: ${o.resultPath})`);
224
- process.exit(o.pass ? 0 : 1);
368
+ console.log(`${o.stats.pass ? "PASS" : "FAIL"} ${o.name} ${formatStats(o.stats)} (results: ${o.resultPath})`);
369
+ process.exit(o.stats.pass ? 0 : 1);
225
370
  }
226
371
  function cmdSweep(argv) {
227
372
  const { positional, flags } = parseArgs(argv);
228
- if (positional.length > 0) fail("usage: skillcheck sweep [--root DIR] [--all]");
373
+ if (positional.length > 0) fail(`usage: skillcheck sweep ${RUN_FLAGS} [--all]`);
229
374
  const root = resolveRoot(flags);
230
375
  const opts = runOptions(flags);
231
376
  ensureEvalPackages(opts);
232
377
  const all = flags.get("--all") === true;
233
378
  const resultsDir = stateDirs(root).results;
379
+ const wanted = configKey(runConfigOf(opts));
234
380
  let passed = 0, failed = 0, errored = 0, skipped = 0;
235
381
  for (const dir of discoverScenarios(root)) {
236
382
  const name = runNameFor(dir, opts.harness);
237
383
  const resultPath = path.join(resultsDir, `${name}.json`);
238
384
  if (!all && fs.existsSync(resultPath) && !fs.existsSync(attemptPath(resultPath))) {
239
- skipped++;
240
- console.log(`SKIP ${name} (results exist; use --all to rerun)`);
241
- continue;
385
+ const raw = readJson(resultPath);
386
+ const config = resultRunConfig(raw, readJson(metaPath(resultPath)), opts.harness);
387
+ const graded = !("error" in classifyResult(raw, config.trials));
388
+ if (graded && configKey(config) === wanted) {
389
+ skipped++;
390
+ console.log(`SKIP ${name} (results exist; use --all to rerun)`);
391
+ continue;
392
+ }
393
+ if (graded) console.log(`RERUN ${name} (results used ${describeConfig(config)})`);
242
394
  }
243
395
  const o = runScenario(dir, opts, root);
244
- if (o.score === void 0) {
396
+ if (o.stats === void 0) {
245
397
  errored++;
246
398
  console.log(`ERROR ${o.name} ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
247
- } else if (o.pass) {
399
+ } else if (o.stats.pass) {
248
400
  passed++;
249
- console.log(`PASS ${o.name} score=${o.score.toFixed(4)}`);
401
+ console.log(`PASS ${o.name} ${formatStats(o.stats)}`);
250
402
  } else {
251
403
  failed++;
252
- console.log(`FAIL ${o.name} score=${o.score.toFixed(4)}`);
404
+ console.log(`FAIL ${o.name} ${formatStats(o.stats)}`);
253
405
  }
254
406
  }
255
407
  console.log(`\nsweep: ${passed} passed, ${failed} failed, ${errored} errored, ${skipped} skipped`);
256
408
  process.exit(errored > 0 ? 2 : failed > 0 ? 1 : 0);
257
409
  }
258
- function resultIdentity(file) {
410
+ function resultIdentity(file, dir) {
411
+ const base = file.replace(/\.json$/, "");
412
+ for (const sidecar of [`${base}.meta.json`, `${file}.attempt`]) try {
413
+ const identity = JSON.parse(fs.readFileSync(path.join(dir, sidecar), "utf8"));
414
+ if (identity !== null && typeof identity === "object" && "skill" in identity && typeof identity.skill === "string" && "scenario" in identity && typeof identity.scenario === "string" && "harness" in identity && (identity.harness === "claude" || identity.harness === "codex" || identity.harness === "grok")) return {
415
+ skill: identity.skill,
416
+ scenario: identity.scenario,
417
+ harness: identity.harness
418
+ };
419
+ } catch {}
420
+ if (/^~v3~[0-9a-f]{64}$/.test(base)) throw new Error(`missing identity metadata for ${file}`);
259
421
  const decode = (part) => {
260
422
  if (!part.startsWith("~v2~")) return part;
261
423
  try {
262
- return decodeURIComponent(part.slice(4));
424
+ const decoded = decodeURIComponent(part.slice(4));
425
+ return encodeRunNamePart(decoded) === part ? decoded : part;
263
426
  } catch {
264
427
  return part;
265
428
  }
266
429
  };
267
- const base = file.replace(/\.json$/, "");
268
430
  const suffix = base.match(/--(codex|grok|cursor)$/);
269
431
  const harness = suffix === null ? "claude" : suffix[1];
270
432
  const [skill, ...rest] = base.replace(/--(codex|grok|cursor)$/, "").split("--");
@@ -277,6 +439,7 @@ function resultIdentity(file) {
277
439
  function reduceResults(dir, allowMixed) {
278
440
  const entries = [];
279
441
  const skipped = [];
442
+ const gradedAt = /* @__PURE__ */ new Map();
280
443
  const shas = /* @__PURE__ */ new Set();
281
444
  const files = fs.readdirSync(dir);
282
445
  const incomplete = new Set(files.filter((f) => f.endsWith(".json.attempt")).map((f) => f.replace(/\.attempt$/, "")));
@@ -292,46 +455,44 @@ function reduceResults(dir, allowMixed) {
292
455
  skipped.push(f);
293
456
  continue;
294
457
  }
295
- let raw;
296
- try {
297
- raw = JSON.parse(fs.readFileSync(path.join(dir, f), "utf8"));
298
- } catch {
299
- raw = void 0;
300
- }
301
- const res = raw?.results?.results?.[0];
302
- const verdict = classifyResult(raw);
303
- if (verdict.score === void 0 || verdict.pass === void 0) {
458
+ const raw = readJson(path.join(dir, f));
459
+ const base = f.replace(/\.json$/, "");
460
+ const meta = readJson(path.join(dir, `${base}.meta.json`));
461
+ const { skill, scenario, harness } = resultIdentity(f, dir);
462
+ const config = resultRunConfig(raw, meta, harness);
463
+ const verdict = classifyResult(raw, config.trials);
464
+ if ("error" in verdict) {
304
465
  console.error(`skipping ${f}: ${verdict.error}`);
305
466
  skipped.push(f);
306
467
  continue;
307
468
  }
308
- const provider = raw.config?.providers?.[0];
309
- const judge = raw.config?.defaultTest?.options?.provider;
310
- const base = f.replace(/\.json$/, "");
311
- const { skill, scenario, harness } = resultIdentity(f);
312
- let sha = "unattested";
313
- try {
314
- sha = JSON.parse(fs.readFileSync(path.join(dir, `${base}.meta.json`), "utf8")).skills_tree_sha ?? "unattested";
315
- } catch {}
469
+ const rows = raw.results.results;
470
+ const key = entryKey({
471
+ skill,
472
+ scenario,
473
+ harness
474
+ });
475
+ gradedAt.set(key, Math.max(gradedAt.get(key) ?? 0, fs.statSync(path.join(dir, f)).mtimeMs));
476
+ const sha = typeof meta?.skills_tree_sha === "string" ? meta.skills_tree_sha : "unattested";
316
477
  shas.add(sha);
478
+ const stats = aggregateTrials(verdict.trials);
317
479
  entries.push({
318
480
  skill,
319
481
  scenario,
320
482
  harness,
321
483
  skills_tree_sha: sha,
322
- score: verdict.score,
323
- pass: verdict.pass,
324
- agent_model: provider?.config?.model ?? `${harness}-default`,
325
- judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? judge?.id ?? "unknown",
326
- latency_ms: res.latencyMs,
327
- tokens: (res.tokenUsage?.total ?? 0) + (res.tokenUsage?.assertions?.total ?? 0)
484
+ ...stats,
485
+ ...config,
486
+ latency_ms: Math.round(rows.reduce((a, r) => a + (r.latencyMs ?? 0), 0) / rows.length),
487
+ tokens: rows.reduce((a, r) => a + (r.tokenUsage?.total ?? 0) + (r.tokenUsage?.assertions?.total ?? 0), 0)
328
488
  });
329
489
  }
330
490
  if (shas.size > 1 && !allowMixed) throw new Error(`results span multiple skills-tree revisions (${[...shas].join(", ")}); rerun stale ones or pass --allow-mixed`);
331
491
  return {
332
492
  treeSha: shas.size === 1 ? [...shas][0] : shas.size === 0 ? "none" : "mixed",
333
493
  entries,
334
- skipped
494
+ skipped,
495
+ gradedAt
335
496
  };
336
497
  }
337
498
  function entryKey(e) {
@@ -375,15 +536,20 @@ function cmdSummarize(argv) {
375
536
  if (positional.length > 0) fail("usage: skillcheck summarize [--root DIR] [--allow-mixed]");
376
537
  const dirs = stateDirs(resolveRoot(flags));
377
538
  if (!fs.existsSync(dirs.results)) fail(`no results directory at ${dirs.results}; run some evals first`);
378
- const { entries, skipped } = reduceResults(dirs.results, flags.get("--allow-mixed") === true);
539
+ const { entries, skipped, gradedAt } = reduceResults(dirs.results, flags.get("--allow-mixed") === true);
379
540
  fs.mkdirSync(dirs.scorecards, { recursive: true });
380
541
  const out = path.join(dirs.scorecards, `${(/* @__PURE__ */ new Date()).toISOString().slice(0, 10)}.json`);
381
542
  const existing = readExistingScorecard(out);
382
- const skippedKeys = new Set(skipped.map((file) => entryKey(resultIdentity(file))));
543
+ const skippedKeys = new Set(skipped.map((file) => ({
544
+ key: entryKey(resultIdentity(file, dirs.results)),
545
+ modifiedAt: fs.statSync(fs.existsSync(path.join(dirs.results, `${file}.attempt`)) ? path.join(dirs.results, `${file}.attempt`) : path.join(dirs.results, file)).mtimeMs
546
+ })).filter(({ key, modifiedAt }) => (gradedAt.get(key) ?? 0) <= modifiedAt).map(({ key }) => key));
383
547
  if (existing.some((entry) => skippedKeys.has(entryKey(entry)))) throw new Error("skipped rerun matches an existing score; refusing to carry it or overwrite the scorecard");
384
548
  const merged = mergeScorecard(existing, entries);
385
549
  const treeSha = treeShaOf(merged.entries);
386
- if (treeSha === "mixed" && flags.get("--allow-mixed") !== true) throw new Error("scorecard spans multiple skills-tree revisions; rerun stale ones or pass --allow-mixed");
550
+ const allowMixed = flags.get("--allow-mixed") === true;
551
+ if (treeSha === "mixed" && !allowMixed) throw new Error("scorecard spans multiple skills-tree revisions; rerun stale ones or pass --allow-mixed");
552
+ assertUniformConfig(merged.entries, allowMixed);
387
553
  const scorecard = {
388
554
  ran_at: (/* @__PURE__ */ new Date()).toISOString(),
389
555
  skills_tree_sha: treeSha,
@@ -392,6 +558,46 @@ function cmdSummarize(argv) {
392
558
  fs.writeFileSync(out, JSON.stringify(scorecard, null, 2) + "\n");
393
559
  console.log(`${out}: ${merged.entries.length} scenario(s), ${merged.entries.filter((e) => e.pass).length} passing, ${skipped.length} skipped file(s)`);
394
560
  if (existing.length > 0) console.log(`merged into today's scorecard: ${entries.length} from this run, ${merged.carried} carried over`);
561
+ console.log(`\n${formatSkillTable(summarizeSkills(merged.entries))}`);
562
+ for (const e of merged.entries.filter((x) => x.noisy)) console.log(`NOISY ${e.skill}/${e.scenario} (${e.harness}) ${formatStats(e)}`);
563
+ }
564
+ function summarizeSkills(entries) {
565
+ const groups = /* @__PURE__ */ new Map();
566
+ for (const e of entries) {
567
+ const key = `${e.skill}\0${e.harness}`;
568
+ groups.set(key, [...groups.get(key) ?? [], e]);
569
+ }
570
+ const mean = (xs) => round4(xs.reduce((a, b) => a + b, 0) / xs.length);
571
+ return [...groups.values()].map((rows) => ({
572
+ skill: rows[0].skill,
573
+ harness: rows[0].harness,
574
+ scenarios: rows.length,
575
+ pass_all: rows.filter((r) => r.pass).length,
576
+ pass_rate: mean(rows.map((r) => r.pass_rate ?? (r.pass ? 1 : 0))),
577
+ score: mean(rows.map((r) => r.score)),
578
+ noisy: rows.filter((r) => r.noisy).map((r) => r.scenario)
579
+ }));
580
+ }
581
+ function formatSkillTable(rows) {
582
+ const table = [[
583
+ "skill",
584
+ "harness",
585
+ "scenarios",
586
+ "pass^k",
587
+ "pass rate",
588
+ "score",
589
+ "noisy"
590
+ ], ...rows.map((r) => [
591
+ r.skill,
592
+ r.harness,
593
+ String(r.scenarios),
594
+ `${r.pass_all}/${r.scenarios}`,
595
+ r.pass_rate.toFixed(2),
596
+ r.score.toFixed(2),
597
+ String(r.noisy.length)
598
+ ])];
599
+ const widths = table[0].map((_, i) => Math.max(...table.map((row) => row[i].length)));
600
+ return table.map((row) => row.map((c, i) => c.padEnd(widths[i])).join(" ").trimEnd()).join("\n");
395
601
  }
396
602
  function cmdLint(argv) {
397
603
  const { positional, flags } = parseArgs(argv);
@@ -427,4 +633,4 @@ if (isMainModule()) {
427
633
  }
428
634
  }
429
635
  //#endregion
430
- export { classifyResult, mergeScorecard, parseArgs, parseMaxTurns, reduceResults, resolveRoot, stateDirs, toolVersion, treeShaOf };
636
+ export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, formatStats, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
package/dist/scenario.js CHANGED
@@ -4,6 +4,7 @@ import path from "node:path";
4
4
  import { createHash } from "node:crypto";
5
5
  import os from "node:os";
6
6
  //#region src/scenario.ts
7
+ const DEFAULT_CLAUDE_AGENT = "claude-opus-5";
7
8
  function loadScenario(scenarioDir) {
8
9
  const match = scenarioDir.match(/skills\/([^/]+)\/evals\/([^/]+)$/);
9
10
  if (!match) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
@@ -40,7 +41,13 @@ function runNameFor(scenarioDir, harness) {
40
41
  const m = path.resolve(scenarioDir).match(/skills\/([^/]+)\/evals\/([^/]+)$/);
41
42
  if (!m) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
42
43
  const name = `${encodeRunNamePart(m[1])}--${encodeRunNamePart(m[2])}`;
43
- return harness === "claude" ? name : `${name}--${harness}`;
44
+ const full = harness === "claude" ? name : `${name}--${harness}`;
45
+ if (Buffer.byteLength(`${full}.json.attempt`) <= 255) return full;
46
+ return `~v3~${createHash("sha256").update(JSON.stringify([
47
+ m[1],
48
+ m[2],
49
+ harness
50
+ ])).digest("hex")}`;
44
51
  }
45
52
  function frontmatterRange(text) {
46
53
  const lines = text.split("\n");
@@ -140,6 +147,7 @@ function agentProvider(opts, workdir, skill, paths) {
140
147
  id: "anthropic:claude-agent-sdk",
141
148
  config: {
142
149
  model: opts.agentModel ?? "claude-opus-5",
150
+ ...opts.agentEffort ? { effort: opts.agentEffort } : {},
143
151
  apiKeyRequired: false,
144
152
  working_dir: workdir,
145
153
  setting_sources: ["project"],
@@ -156,11 +164,17 @@ function agentProvider(opts, workdir, skill, paths) {
156
164
  }
157
165
  };
158
166
  }
159
- function buildConfig(s, workdir, manifestPath, opts, paths) {
167
+ function trialLabel(index) {
168
+ return `trial-${index + 1}`;
169
+ }
170
+ function buildConfig(s, trials, opts, paths) {
160
171
  return {
161
172
  description: `${s.skill}/${s.scenario}`,
162
173
  prompts: ["{{task}}"],
163
- providers: [agentProvider(opts, workdir, s.skill, paths)],
174
+ providers: trials.map((t, i) => ({
175
+ ...agentProvider(opts, t.workdir, s.skill, paths),
176
+ label: trialLabel(i)
177
+ })),
164
178
  defaultTest: { options: {
165
179
  provider: opts.judgeModel.includes(":") ? opts.judgeEffort === void 0 ? opts.judgeModel : {
166
180
  id: opts.judgeModel,
@@ -196,12 +210,13 @@ function buildConfig(s, workdir, manifestPath, opts, paths) {
196
210
  },
197
211
  transform: `file://${paths.transformPath}`
198
212
  } },
199
- tests: [{
213
+ tests: trials.map((t, i) => ({
200
214
  description: s.criteria.context,
215
+ providers: [trialLabel(i)],
201
216
  vars: {
202
217
  task: s.prompt,
203
- workdir,
204
- manifest: manifestPath
218
+ workdir: t.workdir,
219
+ manifest: t.manifestPath
205
220
  },
206
221
  assert: [{
207
222
  type: "assert-set",
@@ -215,7 +230,7 @@ function buildConfig(s, workdir, manifestPath, opts, paths) {
215
230
  type: "skill-used",
216
231
  value: s.skill
217
232
  }]
218
- }]
233
+ }))
219
234
  };
220
235
  }
221
236
  const SDK_PACKAGES = ["@anthropic-ai/claude-agent-sdk", "@openai/codex-sdk"];
@@ -244,7 +259,11 @@ function generateRun(scenarioDir, opts, paths) {
244
259
  const s = loadScenario(scenarioDir);
245
260
  const name = runNameFor(scenarioDir, opts.harness);
246
261
  const runDir = path.join(paths.scratchDir, name);
247
- const { workdir, manifestPath } = materialize(s, runDir, opts.harness);
262
+ fs.rmSync(runDir, {
263
+ recursive: true,
264
+ force: true
265
+ });
266
+ const trials = Array.from({ length: opts.trials ?? 1 }, (_, i) => materialize(s, path.join(runDir, trialLabel(i)), opts.harness));
248
267
  const sdkDir = sdkNodeModulesDir();
249
268
  if (sdkDir !== void 0) {
250
269
  const link = path.join(runDir, "node_modules");
@@ -254,13 +273,15 @@ function generateRun(scenarioDir, opts, paths) {
254
273
  });
255
274
  fs.symlinkSync(sdkDir, link, "dir");
256
275
  }
257
- const config = buildConfig(s, workdir, manifestPath, opts, paths);
276
+ const config = buildConfig(s, trials, opts, paths);
258
277
  const configPath = path.join(runDir, "promptfooconfig.json");
259
278
  fs.writeFileSync(configPath, JSON.stringify(config, null, 2));
260
279
  return {
261
280
  name,
262
- configPath
281
+ configPath,
282
+ skill: s.skill,
283
+ scenario: s.scenario
263
284
  };
264
285
  }
265
286
  //#endregion
266
- export { buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag };
287
+ export { DEFAULT_CLAUDE_AGENT, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
package/docs/adoption.md CHANGED
@@ -66,9 +66,11 @@ Commit `.skillcheck/scorecards/`. Gitignore the rest:
66
66
  .skillcheck/scratch/
67
67
  ```
68
68
 
69
- A scorecard is only comparable against the tree it graded, which is why every
70
- result carries the root repo's HEAD and `summarize` refuses to mix revisions
71
- without `--allow-mixed`.
69
+ A scorecard is only comparable against the tree it graded and the configuration
70
+ it ran with, which is why every result carries the root repo's HEAD, the agent
71
+ and judge models and efforts, and the trial count, and why `summarize` refuses
72
+ to mix them without `--allow-mixed`. Use `--trials 3` or more before calling a
73
+ skill change better or worse; one trial cannot tell a regression from noise.
72
74
 
73
75
  ## Upgrading
74
76
 
package/docs/scenarios.md CHANGED
@@ -54,7 +54,9 @@ item needs a non-empty `name` and `description` and a positive `max_score`.
54
54
  Each item becomes one `llm-rubric` assertion weighted by `max_score`, inside an
55
55
  assert-set with threshold 0.7. A separate `skill-used` assertion sits outside
56
56
  that aggregate, so a run that produces good output without ever loading the
57
- skill still fails. There is no test-level threshold: both must pass.
57
+ skill still fails. There is no test-level threshold: both must pass. The
58
+ reported score is the assert-set's weighted score; skill-used is reported
59
+ separately as a rate across trials.
58
60
 
59
61
  Write descriptions a judge can check against the deliverable: an observable
60
62
  property, not a feeling. Weight the items that would make a reviewer reject the
package/docs/usage.md CHANGED
@@ -35,9 +35,10 @@ skillcheck run skills/<skill>/evals/<scenario>
35
35
  skillcheck run <scenario-dir> --agent MODEL --judge MODEL --harness codex
36
36
  skillcheck run <scenario-dir> --harness claude --max-turns 80
37
37
  skillcheck run <scenario-dir> --harness grok
38
+ skillcheck run <scenario-dir> --trials 3 --agent-effort medium
38
39
  ```
39
40
 
40
- Materializes the scenario into `<root>/.skillcheck/scratch/<name>/workdir`,
41
+ Materializes the scenario into `<root>/.skillcheck/scratch/<name>/trial-<n>/workdir`,
41
42
  installs the skill under test into that workdir, drives the agent, and grades
42
43
  the files it wrote. Exit 0 means pass, 1 means graded fail, and 2 means error.
43
44
  Exit 2 covers missing usable promptfoo output or optional eval peers. The
@@ -53,6 +54,44 @@ and a Claude agent limit of 50 turns. `--max-turns` changes that limit only for
53
54
  Claude; passing it with `codex` or `grok` fails before the eval starts. On
54
55
  those harnesses, omitting `--agent` leaves the model to that CLI's own default.
55
56
 
57
+ ### Trials
58
+
59
+ One trial is one sample of a noisy process: the same scenario and skill can
60
+ score 0.49 and then 0.99. `--trials <k>` (default 1) runs the agent k times,
61
+ each in its own workdir with its own manifest, all graded in one promptfoo eval.
62
+ promptfoo's `--repeat` is not used because it reuses one set of vars and one
63
+ `working_dir`, so concurrent trials would write into the same tree and each
64
+ would be graded on all of their deliverables.
65
+
66
+ A scenario's result aggregates its trials:
67
+
68
+ | Field | Meaning |
69
+ | ------------------------------- | ------------------------------------------------------------------- |
70
+ | `pass` | pass^k: every trial passed. Exit 0 needs this |
71
+ | `passes`, `pass_rate` | Trials that passed, as a count and a fraction |
72
+ | `score` | Mean weighted checklist score (the assert-set, without skill-used) |
73
+ | `score_min` | Lowest trial score |
74
+ | `score_spread` | Highest minus lowest trial score |
75
+ | `skill_used`, `skill_used_rate` | Trials whose `skill-used` assertion passed, as a count and fraction |
76
+ | `noisy` | Trials both passed and failed, or `score_spread` is at least 0.2 |
77
+
78
+ A trial that errored was never graded, so one errored trial makes the whole
79
+ scenario an ERROR: pass^k over fewer than k trials is not the requested number.
80
+ So is a result with fewer rows than trials, or a row without its checklist and
81
+ `skill-used` components.
82
+ With `--trials` above 1, `run` and `sweep` print min, spread, pass count,
83
+ skill-used count, and `NOISY`.
84
+
85
+ ### Agent effort
86
+
87
+ `--agent-effort low|medium|high|xhigh|max` sets the Claude agent's effort.
88
+ promptfoo passes it to the Agent SDK, which starts Claude Code with `--effort`.
89
+ Omitting it leaves Claude Code's default. Only `--harness claude` takes it;
90
+ `codex` and `grok` fail before the eval starts. It is separate from
91
+ `--judge-effort`.
92
+
93
+ ### Harnesses and judges
94
+
56
95
  `--harness grok` runs the locally installed Grok Build CLI in the disposable
57
96
  workdir with the skill under `.grok/skills/`. It uses native streaming events
58
97
  to count a completed read of that skill's `SKILL.md` as `skill-used` evidence.
@@ -86,7 +125,14 @@ order, sequentially. A scenario needs both `task.md` and `criteria.json` to be
86
125
  discovered. Exit 2 if anything errored, 1 if anything failed, else 0.
87
126
 
88
127
  `EVALS_CONCURRENCY` is passed to promptfoo as `-j` (default 4). It parallelizes
89
- within one scenario, not across them.
128
+ the trials of one scenario, not separate scenarios. To spread scenarios, start
129
+ several `skillcheck run` processes; each scenario has its own result, attempt
130
+ marker, and scratch directory.
131
+
132
+ A scenario is skipped only when its completed result was graded with the same
133
+ run configuration: agent model, agent effort, judge model, judge effort, and
134
+ trial count. A result from another configuration is rerun and reported as
135
+ `RERUN`.
90
136
 
91
137
  One known failure mode: judge calls through a gateway can drop at the transport
92
138
  layer ([uinaf/zebroid-infra#44](https://github.com/uinaf/zebroid-infra/issues/44)).
@@ -102,7 +148,10 @@ skillcheck summarize [--allow-mixed]
102
148
 
103
149
  Reduces `<root>/.skillcheck/results/*.json` into
104
150
  `<root>/.skillcheck/scorecards/<UTC-date>.json`: one entry per scenario with
105
- skill, scenario, harness, tree sha, score, pass, both models, latency, tokens.
151
+ skill, scenario, harness, tree sha, the trial aggregate from [trials](#trials),
152
+ the run configuration, mean latency per trial, and tokens summed over trials.
153
+ It then prints one row per skill and harness: scenarios, pass^k count, mean pass
154
+ rate, mean score, and noisy count. Each noisy scenario follows on its own line.
106
155
 
107
156
  If a scorecard for today already exists, the two are merged on
108
157
  `(skill, scenario, harness)`: entries from this run win, entries it did not
@@ -115,14 +164,19 @@ Files that are not promptfoo results and ungraded transport errors are skipped
115
164
  with a warning rather than failing the reduction. Graded assertion failures
116
165
  remain scored results. If a skipped file matches an existing scorecard row,
117
166
  summary generation fails and leaves the scorecard unchanged, so an errored rerun
118
- cannot carry forward its old score. This also applies with `--allow-mixed`.
167
+ cannot carry forward its old score. A graded result for the same identity
168
+ supersedes a skipped attempt only when the result file is newer. This also
169
+ applies with `--allow-mixed`.
119
170
  Results from the retired Cursor harness are skipped with their original identity,
120
171
  so they cannot become Claude scores or silently carry an old Cursor row.
121
172
  Runs keep a `<name>.json.attempt` marker until a graded result and its provenance
122
- are written. An outstanding marker makes `summarize` skip that identity even
173
+ are written. The marker records the original skill, scenario, and harness. An outstanding marker makes `summarize` skip that identity even
123
174
  when the child produced no result file or left partial output. The marker does
124
175
  not count as a result for the sweep's existence check, so no-output failures
125
176
  remain eligible for retry.
177
+ Long escaped names use a short hashed filename; the original identity is kept
178
+ in the attempt marker and result sidecar. `sweep` retries existing results that
179
+ contain no grade, including results written by older versions without a marker.
126
180
 
127
181
  ## Provenance
128
182
 
@@ -131,14 +185,26 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
131
185
  ```json
132
186
  {
133
187
  "skills_tree_sha": "<root repo HEAD at run time>",
188
+ "skill": "<skill directory name>",
189
+ "scenario": "<scenario directory name>",
134
190
  "harness": "claude",
191
+ "agent_model": "claude-opus-5",
192
+ "agent_effort": "medium",
193
+ "judge_model": "claude-opus-5",
194
+ "judge_effort": null,
195
+ "trials": 3,
196
+ "aggregate": { "pass": false, "pass_rate": 0.6667, "score": 0.81, "...": "..." },
135
197
  "ran_at": "<ISO timestamp>",
136
198
  "tool_version": "<skillcheck version>"
137
199
  }
138
200
  ```
139
201
 
140
- `summarize` reads those sidecars and refuses to mix skills-tree revisions in one
141
- scorecard, including retained rows from partial reruns, unless `--allow-mixed`.
202
+ `summarize` reads those sidecars and refuses to mix skills-tree revisions or
203
+ run configurations (agent model and effort, judge model and effort, trials) in
204
+ one scorecard, including retained rows from partial reruns, unless
205
+ `--allow-mixed`. Configurations are compared within a harness, since harnesses
206
+ differ by design. Sidecars written before run configurations were recorded fall
207
+ back to the promptfoo config stored in the result.
142
208
  Rejection leaves the existing scorecard unchanged. With the override, the top-level `skills_tree_sha`
143
209
  becomes `mixed` and per-entry shas remain. A result with no sidecar reduces as
144
210
  `unattested`.
@@ -162,3 +228,10 @@ written inside the installed package.
162
228
 
163
229
  A bare `--judge` model stays on the Anthropic selection regardless of the
164
230
  agent harness; a provider-qualified `--judge` uses that provider's env instead.
231
+
232
+ The Claude agent loads project settings only, so an `apiKeyHelper` or `env`
233
+ block in the operator's `~/.claude/settings.json` never reaches it; the run
234
+ fails with `Not logged in`. Export the gateway variables instead:
235
+ `ANTHROPIC_BASE_URL` and `ANTHROPIC_AUTH_TOKEN` set to the helper's output.
236
+ Running from inside a Claude Code session also leaks that session's
237
+ `CLAUDECODE` and `CLAUDE_CODE_*` variables into the agent; run from a plain shell.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "1.0.1",
3
+ "version": "1.1.0",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {
@@ -33,17 +33,17 @@
33
33
  "prepublishOnly": "pnpm run verify:full"
34
34
  },
35
35
  "devDependencies": {
36
- "@anthropic-ai/claude-agent-sdk": "^0.3.233",
37
- "@openai/codex-sdk": "^0.147.0",
38
- "@types/node": "^26.2.0",
39
- "promptfoo": "^0.122.0",
36
+ "@anthropic-ai/claude-agent-sdk": "^0.3.281",
37
+ "@openai/codex-sdk": "^0.156.1",
38
+ "@types/node": "^26.6.2",
39
+ "promptfoo": "^0.123.1",
40
40
  "vite": "catalog:",
41
41
  "vite-plus": "catalog:"
42
42
  },
43
43
  "peerDependencies": {
44
- "@anthropic-ai/claude-agent-sdk": "^0.3.233",
45
- "@openai/codex-sdk": "^0.147.0",
46
- "promptfoo": "^0.122.0"
44
+ "@anthropic-ai/claude-agent-sdk": "^0.3.281",
45
+ "@openai/codex-sdk": "^0.156.1",
46
+ "promptfoo": "^0.123.1"
47
47
  },
48
48
  "peerDependenciesMeta": {
49
49
  "@anthropic-ai/claude-agent-sdk": {
@@ -56,8 +56,15 @@
56
56
  "optional": true
57
57
  }
58
58
  },
59
+ "devEngines": {
60
+ "runtime": {
61
+ "name": "node",
62
+ "version": "^24.11.0 || >=26.0.0",
63
+ "onFail": "error"
64
+ }
65
+ },
59
66
  "engines": {
60
67
  "node": ">=24"
61
68
  },
62
- "packageManager": "pnpm@12.0.0"
69
+ "packageManager": "pnpm@12.4.2"
63
70
  }