@uinaf/skillcheck 1.0.2 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -28,7 +28,8 @@ pre-npm git tags are covered there too.
28
28
  ```sh
29
29
  skillcheck lint # structural lint, no credentials
30
30
  skillcheck run skills/wat/evals/basic # one scenario, graded end to end
31
- skillcheck sweep && skillcheck summarize # every scenario, then a scorecard
31
+ skillcheck sweep --trials 3 # every scenario, three trials each
32
+ skillcheck summarize # per-scenario and per-skill scorecard
32
33
  ```
33
34
 
34
35
  `lint` needs nothing. `run` and `sweep` need model auth, which is why sweeps
package/dist/cli.js CHANGED
@@ -12,11 +12,18 @@ const selfExt = path.extname(fileURLToPath(import.meta.url));
12
12
  function toolVersion() {
13
13
  return JSON.parse(fs.readFileSync(path.join(packageDir, "package.json"), "utf8")).version;
14
14
  }
15
- function parseMaxTurns(raw) {
15
+ function parsePositiveInt(flag, raw) {
16
16
  const n = Number(raw);
17
- if (!Number.isInteger(n) || n <= 0) throw new Error(`--max-turns must be a positive integer, got ${JSON.stringify(raw)}`);
17
+ if (!Number.isInteger(n) || n <= 0) throw new Error(`${flag} must be a positive integer, got ${JSON.stringify(raw)}`);
18
18
  return n;
19
19
  }
20
+ const AGENT_EFFORTS = [
21
+ "low",
22
+ "medium",
23
+ "high",
24
+ "xhigh",
25
+ "max"
26
+ ];
20
27
  function fail(msg) {
21
28
  console.error(msg);
22
29
  process.exit(1);
@@ -29,8 +36,10 @@ function parseArgs(argv) {
29
36
  "--agent",
30
37
  "--judge",
31
38
  "--judge-effort",
39
+ "--agent-effort",
32
40
  "--harness",
33
- "--max-turns"
41
+ "--max-turns",
42
+ "--trials"
34
43
  ]);
35
44
  for (let i = 0; i < argv.length; i++) {
36
45
  const a = argv[i];
@@ -60,28 +69,67 @@ function stateDirs(root) {
60
69
  }
61
70
  function runOptions(flags) {
62
71
  const harness = flags.get("--harness") ?? "claude";
63
- if (harness !== "claude" && harness !== "codex" && harness !== "grok") fail(`--harness must be claude, codex, or grok, got ${harness}`);
64
- if (flags.has("--max-turns") && harness !== "claude") fail("--max-turns is only supported with --harness claude");
72
+ if (harness !== "claude" && harness !== "codex" && harness !== "grok") throw new Error(`--harness must be claude, codex, or grok, got ${harness}`);
73
+ if (flags.has("--max-turns") && harness !== "claude") throw new Error("--max-turns is only supported with --harness claude");
74
+ const agentEffort = flags.get("--agent-effort");
75
+ if (agentEffort !== void 0) {
76
+ if (harness !== "claude") throw new Error(`--agent-effort is only supported with --harness claude; ${harness} has no effort setting wired`);
77
+ if (!AGENT_EFFORTS.includes(agentEffort)) throw new Error(`--agent-effort must be ${AGENT_EFFORTS.join(", ")}, got ${agentEffort}`);
78
+ }
65
79
  const agent = flags.get("--agent");
66
80
  const judgeModel = flags.get("--judge") ?? "claude-opus-5";
67
81
  const judgeEffort = flags.get("--judge-effort");
68
82
  if (judgeEffort !== void 0) {
69
- if (![
83
+ const levels = judgeModel.includes(":") ? [
70
84
  "minimal",
71
85
  "low",
72
86
  "medium",
73
87
  "high"
74
- ].includes(judgeEffort)) fail(`--judge-effort must be minimal, low, medium, or high, got ${judgeEffort}`);
75
- if (!judgeModel.includes(":")) fail("--judge-effort needs a provider-qualified --judge (e.g. openai:chat:gpt-5.6-sol); the Anthropic judge does not take a reasoning effort");
88
+ ] : AGENT_EFFORTS;
89
+ if (!levels.includes(judgeEffort)) throw new Error(`--judge-effort for ${judgeModel} must be ${levels.join(", ")}, got ${judgeEffort}`);
76
90
  }
77
91
  return {
78
92
  harness,
79
93
  agentModel: agent,
94
+ agentEffort,
80
95
  judgeModel,
81
96
  judgeEffort,
82
- maxTurns: flags.has("--max-turns") ? parseMaxTurns(flags.get("--max-turns")) : void 0
97
+ maxTurns: flags.has("--max-turns") ? parsePositiveInt("--max-turns", flags.get("--max-turns")) : void 0,
98
+ trials: flags.has("--trials") ? parsePositiveInt("--trials", flags.get("--trials")) : 1
99
+ };
100
+ }
101
+ function runConfigOf(opts) {
102
+ return {
103
+ agent_model: opts.agentModel ?? (opts.harness === "claude" ? "claude-opus-5" : `${opts.harness}-default`),
104
+ agent_effort: opts.agentEffort ?? null,
105
+ judge_model: opts.judgeModel,
106
+ judge_effort: opts.judgeEffort ?? null,
107
+ trials: opts.trials ?? 1
83
108
  };
84
109
  }
110
+ function configKey(c) {
111
+ return JSON.stringify([
112
+ c.agent_model,
113
+ c.agent_effort ?? null,
114
+ c.judge_model,
115
+ c.judge_effort ?? null,
116
+ c.trials ?? 1
117
+ ]);
118
+ }
119
+ function describeConfig(c) {
120
+ const effort = (e) => e ? `@${e}` : "";
121
+ return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}`;
122
+ }
123
+ function assertUniformConfig(entries, allowMixed) {
124
+ if (allowMixed) return;
125
+ const byHarness = /* @__PURE__ */ new Map();
126
+ for (const e of entries) {
127
+ const configs = byHarness.get(e.harness) ?? /* @__PURE__ */ new Map();
128
+ configs.set(configKey(e), e);
129
+ byHarness.set(e.harness, configs);
130
+ }
131
+ for (const [harness, configs] of byHarness) if (configs.size > 1) throw new Error(`${harness} results span multiple run configurations (${[...configs.values()].map(describeConfig).join("; ")}); rerun them to match or pass --allow-mixed`);
132
+ }
85
133
  function ensureEvalPackages(opts) {
86
134
  const missing = requiredEvalPackages(opts, process.env.ANTHROPIC_API_KEY !== void 0).filter((pkg) => resolvePackageDir(pkg) === void 0);
87
135
  if (missing.length === 0) return;
@@ -105,21 +153,76 @@ function promptfooEntry() {
105
153
  if (typeof rel !== "string") throw new Error(`promptfoo at ${dir} declares no bin`);
106
154
  return path.join(dir, rel);
107
155
  }
108
- function classifyResult(raw) {
156
+ const GRADED_REASONS = /* @__PURE__ */ new Set([0, 1]);
157
+ const ERRORED_REASON = 2;
158
+ function classifyResult(raw, expectedTrials) {
109
159
  const root = raw;
110
- const res = root?.results?.results?.[0];
160
+ const rows = root?.results?.results;
161
+ if (!Array.isArray(rows) || rows.length === 0) return { error: "promptfoo output carried no result" };
162
+ if (expectedTrials !== void 0 && rows.length !== expectedTrials) return { error: `promptfoo returned ${rows.length} of ${expectedTrials} trials` };
163
+ const stats = rows.length === 1 ? root?.results?.stats ?? void 0 : void 0;
164
+ const trials = [];
165
+ for (const [i, row] of rows.entries()) {
166
+ const verdict = classifyRow(row, stats);
167
+ if ("error" in verdict) return { error: rows.length === 1 ? verdict.error : `trial ${i + 1}: ${verdict.error}` };
168
+ trials.push(verdict);
169
+ }
170
+ return { trials };
171
+ }
172
+ function classifyRow(raw, stats) {
173
+ const res = raw;
111
174
  if (res === void 0 || res === null) return { error: "promptfoo output carried no result" };
112
175
  const message = typeof res.error === "string" ? res.error.trim() : "";
113
- const stats = root?.results?.stats ?? void 0;
114
- const gradedByStats = stats !== void 0 && (stats.errors ?? 0) === 0 && ((stats.failures ?? 0) > 0 || (stats.successes ?? 0) > 0);
115
- if (message !== "" && !gradedByStats) return { error: message };
116
- if (stats !== void 0 && (stats.errors ?? 0) > 0 && (stats.successes ?? 0) === 0 && (stats.failures ?? 0) === 0) return { error: "promptfoo reported an errored test with nothing graded" };
176
+ if (res.failureReason === ERRORED_REASON) return { error: message || "promptfoo reported an errored test" };
177
+ const graded = GRADED_REASONS.has(res.failureReason) || stats !== void 0 && (stats.errors ?? 0) === 0 && ((stats.failures ?? 0) > 0 || (stats.successes ?? 0) > 0);
178
+ if (message !== "" && !graded) return { error: message };
179
+ if (!graded && stats !== void 0 && (stats.errors ?? 0) > 0 && (stats.successes ?? 0) === 0 && (stats.failures ?? 0) === 0) return { error: "promptfoo reported an errored test with nothing graded" };
117
180
  if (typeof res.score !== "number" || typeof res.success !== "boolean") return { error: "promptfoo result carried no usable score" };
181
+ const components = Array.isArray(res.gradingResult?.componentResults) ? res.gradingResult.componentResults : [];
182
+ const checklist = components.find((c) => c?.metadata?.assertionSet?.type === "assert-set");
183
+ const skillUsed = components.find((c) => c?.assertion?.type === "skill-used");
184
+ if (typeof checklist?.score !== "number" || typeof skillUsed?.pass !== "boolean") return { error: "promptfoo result carried no checklist or skill-used verdict" };
118
185
  return {
119
- score: res.score,
120
- pass: res.success
186
+ score: checklist.score,
187
+ pass: res.success,
188
+ skillUsed: skillUsed.pass
121
189
  };
122
190
  }
191
+ const NOISY_SPREAD = .2;
192
+ const round4 = (n) => Math.round(n * 1e4) / 1e4;
193
+ function aggregateTrials(trials) {
194
+ if (trials.length === 0) throw new Error("cannot aggregate zero trials");
195
+ const scores = trials.map((t) => t.score);
196
+ const passes = trials.filter((t) => t.pass).length;
197
+ const min = Math.min(...scores);
198
+ const spread = Math.max(...scores) - min;
199
+ const skillUsed = trials.filter((t) => t.skillUsed).length;
200
+ return {
201
+ trials: trials.length,
202
+ pass: passes === trials.length,
203
+ passes,
204
+ pass_rate: round4(passes / trials.length),
205
+ score: round4(scores.reduce((a, b) => a + b, 0) / trials.length),
206
+ score_min: round4(min),
207
+ score_spread: round4(spread),
208
+ skill_used: skillUsed,
209
+ skill_used_rate: round4(skillUsed / trials.length),
210
+ noisy: passes > 0 && passes < trials.length || spread >= .199999999
211
+ };
212
+ }
213
+ function formatStats(s) {
214
+ const line = `score=${s.score.toFixed(4)}`;
215
+ if (s.trials === 1) return line;
216
+ return [
217
+ line,
218
+ `min=${s.score_min.toFixed(4)}`,
219
+ `spread=${s.score_spread.toFixed(4)}`,
220
+ `pass^${s.trials}=${s.pass ? "yes" : "no"}`,
221
+ `passes=${s.passes}/${s.trials}`,
222
+ `skill-used=${s.skill_used}/${s.trials}`,
223
+ ...s.noisy ? ["NOISY"] : []
224
+ ].join(" ");
225
+ }
123
226
  function gitHead(root) {
124
227
  return execFileSync("git", ["rev-parse", "HEAD"], {
125
228
  cwd: root,
@@ -177,24 +280,62 @@ function runScenario(scenarioDir, opts, root) {
177
280
  if (rc !== 0) return outcome;
178
281
  let verdict;
179
282
  try {
180
- verdict = classifyResult(JSON.parse(fs.readFileSync(resultPath, "utf8")));
283
+ verdict = classifyResult(JSON.parse(fs.readFileSync(resultPath, "utf8")), opts.trials ?? 1);
181
284
  } catch {
182
285
  verdict = { error: "promptfoo produced no parseable result file" };
183
286
  }
184
- outcome.score = verdict.score;
185
- outcome.pass = verdict.pass;
186
- outcome.error = verdict.error;
187
- if (verdict.score !== void 0) {
188
- fs.writeFileSync(metaPath(resultPath), JSON.stringify({
189
- skills_tree_sha: sha,
190
- ...identity,
191
- ran_at: (/* @__PURE__ */ new Date()).toISOString(),
192
- tool_version: toolVersion()
193
- }, null, 2) + "\n");
194
- fs.rmSync(attemptPath(resultPath), { force: true });
195
- }
287
+ if ("error" in verdict) return {
288
+ ...outcome,
289
+ error: verdict.error
290
+ };
291
+ outcome.stats = aggregateTrials(verdict.trials);
292
+ fs.writeFileSync(metaPath(resultPath), JSON.stringify({
293
+ skills_tree_sha: sha,
294
+ ...identity,
295
+ ...runConfigOf(opts),
296
+ aggregate: outcome.stats,
297
+ ran_at: (/* @__PURE__ */ new Date()).toISOString(),
298
+ tool_version: toolVersion()
299
+ }, null, 2) + "\n");
300
+ fs.rmSync(attemptPath(resultPath), { force: true });
196
301
  return outcome;
197
302
  }
303
+ function judgeName(judge) {
304
+ if (typeof judge === "string") return judge.replace(/^anthropic:messages:/, "");
305
+ const j = judge;
306
+ const name = j?.config?.model ?? j?.id;
307
+ if (typeof name !== "string") return "unknown";
308
+ return j?.config?.effort === void 0 ? name : name.replace(/^anthropic:messages:/, "");
309
+ }
310
+ function resultRunConfig(raw, meta, harness) {
311
+ const m = meta;
312
+ if (typeof m?.agent_model === "string" && typeof m.judge_model === "string") return {
313
+ agent_model: m.agent_model,
314
+ agent_effort: m.agent_effort ?? null,
315
+ judge_model: m.judge_model,
316
+ judge_effort: m.judge_effort ?? null,
317
+ trials: m.trials ?? 1
318
+ };
319
+ const r = raw;
320
+ const agent = r?.config?.providers?.[0]?.config;
321
+ const judge = r?.config?.defaultTest?.options?.provider;
322
+ const judgeConfig = judge?.config;
323
+ const judgeEffort = judgeConfig?.reasoning_effort ?? judgeConfig?.effort;
324
+ return {
325
+ agent_model: typeof agent?.model === "string" ? agent.model : `${harness}-default`,
326
+ agent_effort: typeof agent?.effort === "string" ? agent.effort : null,
327
+ judge_model: judgeName(judge),
328
+ judge_effort: typeof judgeEffort === "string" ? judgeEffort : null,
329
+ trials: r?.results?.results?.length ?? 1
330
+ };
331
+ }
332
+ function readJson(file) {
333
+ try {
334
+ return JSON.parse(fs.readFileSync(file, "utf8"));
335
+ } catch {
336
+ return;
337
+ }
338
+ }
198
339
  function discoverScenarios(root) {
199
340
  const roots = [path.join(root, "skills")];
200
341
  const cliDir = path.join(root, "cli");
@@ -215,48 +356,54 @@ function discoverScenarios(root) {
215
356
  }
216
357
  return found.sort();
217
358
  }
359
+ const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N]";
218
360
  function cmdRun(argv) {
219
361
  const { positional, flags } = parseArgs(argv);
220
- if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--judge-effort EFFORT] [--harness claude|codex|grok]");
362
+ if (positional.length !== 1) fail(`usage: skillcheck run <scenario-dir> ${RUN_FLAGS}`);
221
363
  const opts = runOptions(flags);
222
364
  ensureEvalPackages(opts);
223
365
  const o = runScenario(positional[0], opts, resolveRoot(flags));
224
- if (o.score === void 0) {
366
+ if (o.stats === void 0) {
225
367
  console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
226
368
  process.exit(2);
227
369
  }
228
- console.log(`${o.pass ? "PASS" : "FAIL"} ${o.name} score=${o.score.toFixed(4)} (results: ${o.resultPath})`);
229
- process.exit(o.pass ? 0 : 1);
370
+ console.log(`${o.stats.pass ? "PASS" : "FAIL"} ${o.name} ${formatStats(o.stats)} (results: ${o.resultPath})`);
371
+ process.exit(o.stats.pass ? 0 : 1);
230
372
  }
231
373
  function cmdSweep(argv) {
232
374
  const { positional, flags } = parseArgs(argv);
233
- if (positional.length > 0) fail("usage: skillcheck sweep [--root DIR] [--all]");
375
+ if (positional.length > 0) fail(`usage: skillcheck sweep ${RUN_FLAGS} [--all]`);
234
376
  const root = resolveRoot(flags);
235
377
  const opts = runOptions(flags);
236
378
  ensureEvalPackages(opts);
237
379
  const all = flags.get("--all") === true;
238
380
  const resultsDir = stateDirs(root).results;
381
+ const wanted = configKey(runConfigOf(opts));
239
382
  let passed = 0, failed = 0, errored = 0, skipped = 0;
240
383
  for (const dir of discoverScenarios(root)) {
241
384
  const name = runNameFor(dir, opts.harness);
242
385
  const resultPath = path.join(resultsDir, `${name}.json`);
243
- if (!all && fs.existsSync(resultPath) && !fs.existsSync(attemptPath(resultPath))) try {
244
- if (classifyResult(JSON.parse(fs.readFileSync(resultPath, "utf8"))).score !== void 0) {
386
+ if (!all && fs.existsSync(resultPath) && !fs.existsSync(attemptPath(resultPath))) {
387
+ const raw = readJson(resultPath);
388
+ const config = resultRunConfig(raw, readJson(metaPath(resultPath)), opts.harness);
389
+ const graded = !("error" in classifyResult(raw, config.trials));
390
+ if (graded && configKey(config) === wanted) {
245
391
  skipped++;
246
392
  console.log(`SKIP ${name} (results exist; use --all to rerun)`);
247
393
  continue;
248
394
  }
249
- } catch {}
395
+ if (graded) console.log(`RERUN ${name} (results used ${describeConfig(config)})`);
396
+ }
250
397
  const o = runScenario(dir, opts, root);
251
- if (o.score === void 0) {
398
+ if (o.stats === void 0) {
252
399
  errored++;
253
400
  console.log(`ERROR ${o.name} ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
254
- } else if (o.pass) {
401
+ } else if (o.stats.pass) {
255
402
  passed++;
256
- console.log(`PASS ${o.name} score=${o.score.toFixed(4)}`);
403
+ console.log(`PASS ${o.name} ${formatStats(o.stats)}`);
257
404
  } else {
258
405
  failed++;
259
- console.log(`FAIL ${o.name} score=${o.score.toFixed(4)}`);
406
+ console.log(`FAIL ${o.name} ${formatStats(o.stats)}`);
260
407
  }
261
408
  }
262
409
  console.log(`\nsweep: ${passed} passed, ${failed} failed, ${errored} errored, ${skipped} skipped`);
@@ -310,45 +457,36 @@ function reduceResults(dir, allowMixed) {
310
457
  skipped.push(f);
311
458
  continue;
312
459
  }
313
- let raw;
314
- try {
315
- raw = JSON.parse(fs.readFileSync(path.join(dir, f), "utf8"));
316
- } catch {
317
- raw = void 0;
318
- }
319
- const res = raw?.results?.results?.[0];
320
- const verdict = classifyResult(raw);
321
- if (verdict.score === void 0 || verdict.pass === void 0) {
460
+ const raw = readJson(path.join(dir, f));
461
+ const base = f.replace(/\.json$/, "");
462
+ const meta = readJson(path.join(dir, `${base}.meta.json`));
463
+ const { skill, scenario, harness } = resultIdentity(f, dir);
464
+ const config = resultRunConfig(raw, meta, harness);
465
+ const verdict = classifyResult(raw, config.trials);
466
+ if ("error" in verdict) {
322
467
  console.error(`skipping ${f}: ${verdict.error}`);
323
468
  skipped.push(f);
324
469
  continue;
325
470
  }
326
- const provider = raw.config?.providers?.[0];
327
- const judge = raw.config?.defaultTest?.options?.provider;
328
- const base = f.replace(/\.json$/, "");
329
- const { skill, scenario, harness } = resultIdentity(f, dir);
471
+ const rows = raw.results.results;
330
472
  const key = entryKey({
331
473
  skill,
332
474
  scenario,
333
475
  harness
334
476
  });
335
477
  gradedAt.set(key, Math.max(gradedAt.get(key) ?? 0, fs.statSync(path.join(dir, f)).mtimeMs));
336
- let sha = "unattested";
337
- try {
338
- sha = JSON.parse(fs.readFileSync(path.join(dir, `${base}.meta.json`), "utf8")).skills_tree_sha ?? "unattested";
339
- } catch {}
478
+ const sha = typeof meta?.skills_tree_sha === "string" ? meta.skills_tree_sha : "unattested";
340
479
  shas.add(sha);
480
+ const stats = aggregateTrials(verdict.trials);
341
481
  entries.push({
342
482
  skill,
343
483
  scenario,
344
484
  harness,
345
485
  skills_tree_sha: sha,
346
- score: verdict.score,
347
- pass: verdict.pass,
348
- agent_model: provider?.config?.model ?? `${harness}-default`,
349
- judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? judge?.id ?? "unknown",
350
- latency_ms: res.latencyMs,
351
- tokens: (res.tokenUsage?.total ?? 0) + (res.tokenUsage?.assertions?.total ?? 0)
486
+ ...stats,
487
+ ...config,
488
+ latency_ms: Math.round(rows.reduce((a, r) => a + (r.latencyMs ?? 0), 0) / rows.length),
489
+ tokens: rows.reduce((a, r) => a + (r.tokenUsage?.total ?? 0) + (r.tokenUsage?.assertions?.total ?? 0), 0)
352
490
  });
353
491
  }
354
492
  if (shas.size > 1 && !allowMixed) throw new Error(`results span multiple skills-tree revisions (${[...shas].join(", ")}); rerun stale ones or pass --allow-mixed`);
@@ -411,7 +549,9 @@ function cmdSummarize(argv) {
411
549
  if (existing.some((entry) => skippedKeys.has(entryKey(entry)))) throw new Error("skipped rerun matches an existing score; refusing to carry it or overwrite the scorecard");
412
550
  const merged = mergeScorecard(existing, entries);
413
551
  const treeSha = treeShaOf(merged.entries);
414
- if (treeSha === "mixed" && flags.get("--allow-mixed") !== true) throw new Error("scorecard spans multiple skills-tree revisions; rerun stale ones or pass --allow-mixed");
552
+ const allowMixed = flags.get("--allow-mixed") === true;
553
+ if (treeSha === "mixed" && !allowMixed) throw new Error("scorecard spans multiple skills-tree revisions; rerun stale ones or pass --allow-mixed");
554
+ assertUniformConfig(merged.entries, allowMixed);
415
555
  const scorecard = {
416
556
  ran_at: (/* @__PURE__ */ new Date()).toISOString(),
417
557
  skills_tree_sha: treeSha,
@@ -420,6 +560,46 @@ function cmdSummarize(argv) {
420
560
  fs.writeFileSync(out, JSON.stringify(scorecard, null, 2) + "\n");
421
561
  console.log(`${out}: ${merged.entries.length} scenario(s), ${merged.entries.filter((e) => e.pass).length} passing, ${skipped.length} skipped file(s)`);
422
562
  if (existing.length > 0) console.log(`merged into today's scorecard: ${entries.length} from this run, ${merged.carried} carried over`);
563
+ console.log(`\n${formatSkillTable(summarizeSkills(merged.entries))}`);
564
+ for (const e of merged.entries.filter((x) => x.noisy)) console.log(`NOISY ${e.skill}/${e.scenario} (${e.harness}) ${formatStats(e)}`);
565
+ }
566
+ function summarizeSkills(entries) {
567
+ const groups = /* @__PURE__ */ new Map();
568
+ for (const e of entries) {
569
+ const key = `${e.skill}\0${e.harness}`;
570
+ groups.set(key, [...groups.get(key) ?? [], e]);
571
+ }
572
+ const mean = (xs) => round4(xs.reduce((a, b) => a + b, 0) / xs.length);
573
+ return [...groups.values()].map((rows) => ({
574
+ skill: rows[0].skill,
575
+ harness: rows[0].harness,
576
+ scenarios: rows.length,
577
+ pass_all: rows.filter((r) => r.pass).length,
578
+ pass_rate: mean(rows.map((r) => r.pass_rate ?? (r.pass ? 1 : 0))),
579
+ score: mean(rows.map((r) => r.score)),
580
+ noisy: rows.filter((r) => r.noisy).map((r) => r.scenario)
581
+ }));
582
+ }
583
+ function formatSkillTable(rows) {
584
+ const table = [[
585
+ "skill",
586
+ "harness",
587
+ "scenarios",
588
+ "pass^k",
589
+ "pass rate",
590
+ "score",
591
+ "noisy"
592
+ ], ...rows.map((r) => [
593
+ r.skill,
594
+ r.harness,
595
+ String(r.scenarios),
596
+ `${r.pass_all}/${r.scenarios}`,
597
+ r.pass_rate.toFixed(2),
598
+ r.score.toFixed(2),
599
+ String(r.noisy.length)
600
+ ])];
601
+ const widths = table[0].map((_, i) => Math.max(...table.map((row) => row[i].length)));
602
+ return table.map((row) => row.map((c, i) => c.padEnd(widths[i])).join(" ").trimEnd()).join("\n");
423
603
  }
424
604
  function cmdLint(argv) {
425
605
  const { positional, flags } = parseArgs(argv);
@@ -455,4 +635,4 @@ if (isMainModule()) {
455
635
  }
456
636
  }
457
637
  //#endregion
458
- export { classifyResult, mergeScorecard, parseArgs, parseMaxTurns, reduceResults, resolveRoot, stateDirs, toolVersion, treeShaOf };
638
+ export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, formatStats, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
package/dist/scenario.js CHANGED
@@ -4,6 +4,7 @@ import path from "node:path";
4
4
  import { createHash } from "node:crypto";
5
5
  import os from "node:os";
6
6
  //#region src/scenario.ts
7
+ const DEFAULT_CLAUDE_AGENT = "claude-opus-5";
7
8
  function loadScenario(scenarioDir) {
8
9
  const match = scenarioDir.match(/skills\/([^/]+)\/evals\/([^/]+)$/);
9
10
  if (!match) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
@@ -146,6 +147,7 @@ function agentProvider(opts, workdir, skill, paths) {
146
147
  id: "anthropic:claude-agent-sdk",
147
148
  config: {
148
149
  model: opts.agentModel ?? "claude-opus-5",
150
+ ...opts.agentEffort ? { effort: opts.agentEffort } : {},
149
151
  apiKeyRequired: false,
150
152
  working_dir: workdir,
151
153
  setting_sources: ["project"],
@@ -162,19 +164,29 @@ function agentProvider(opts, workdir, skill, paths) {
162
164
  }
163
165
  };
164
166
  }
165
- function buildConfig(s, workdir, manifestPath, opts, paths) {
167
+ function trialLabel(index) {
168
+ return `trial-${index + 1}`;
169
+ }
170
+ function buildConfig(s, trials, opts, paths) {
166
171
  return {
167
172
  description: `${s.skill}/${s.scenario}`,
168
173
  prompts: ["{{task}}"],
169
- providers: [agentProvider(opts, workdir, s.skill, paths)],
174
+ providers: trials.map((t, i) => ({
175
+ ...agentProvider(opts, t.workdir, s.skill, paths),
176
+ label: trialLabel(i)
177
+ })),
170
178
  defaultTest: { options: {
171
179
  provider: opts.judgeModel.includes(":") ? opts.judgeEffort === void 0 ? opts.judgeModel : {
172
180
  id: opts.judgeModel,
173
181
  config: { reasoning_effort: opts.judgeEffort }
174
- } : process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
182
+ } : process.env.ANTHROPIC_API_KEY ? opts.judgeEffort === void 0 ? `anthropic:messages:${opts.judgeModel}` : {
183
+ id: `anthropic:messages:${opts.judgeModel}`,
184
+ config: { effort: opts.judgeEffort }
185
+ } : {
175
186
  id: "anthropic:claude-agent-sdk",
176
187
  config: {
177
188
  model: opts.judgeModel,
189
+ ...opts.judgeEffort ? { effort: opts.judgeEffort } : {},
178
190
  apiKeyRequired: false,
179
191
  max_turns: 3,
180
192
  output_format: {
@@ -202,12 +214,13 @@ function buildConfig(s, workdir, manifestPath, opts, paths) {
202
214
  },
203
215
  transform: `file://${paths.transformPath}`
204
216
  } },
205
- tests: [{
217
+ tests: trials.map((t, i) => ({
206
218
  description: s.criteria.context,
219
+ providers: [trialLabel(i)],
207
220
  vars: {
208
221
  task: s.prompt,
209
- workdir,
210
- manifest: manifestPath
222
+ workdir: t.workdir,
223
+ manifest: t.manifestPath
211
224
  },
212
225
  assert: [{
213
226
  type: "assert-set",
@@ -221,7 +234,7 @@ function buildConfig(s, workdir, manifestPath, opts, paths) {
221
234
  type: "skill-used",
222
235
  value: s.skill
223
236
  }]
224
- }]
237
+ }))
225
238
  };
226
239
  }
227
240
  const SDK_PACKAGES = ["@anthropic-ai/claude-agent-sdk", "@openai/codex-sdk"];
@@ -250,7 +263,11 @@ function generateRun(scenarioDir, opts, paths) {
250
263
  const s = loadScenario(scenarioDir);
251
264
  const name = runNameFor(scenarioDir, opts.harness);
252
265
  const runDir = path.join(paths.scratchDir, name);
253
- const { workdir, manifestPath } = materialize(s, runDir, opts.harness);
266
+ fs.rmSync(runDir, {
267
+ recursive: true,
268
+ force: true
269
+ });
270
+ const trials = Array.from({ length: opts.trials ?? 1 }, (_, i) => materialize(s, path.join(runDir, trialLabel(i)), opts.harness));
254
271
  const sdkDir = sdkNodeModulesDir();
255
272
  if (sdkDir !== void 0) {
256
273
  const link = path.join(runDir, "node_modules");
@@ -260,7 +277,7 @@ function generateRun(scenarioDir, opts, paths) {
260
277
  });
261
278
  fs.symlinkSync(sdkDir, link, "dir");
262
279
  }
263
- const config = buildConfig(s, workdir, manifestPath, opts, paths);
280
+ const config = buildConfig(s, trials, opts, paths);
264
281
  const configPath = path.join(runDir, "promptfooconfig.json");
265
282
  fs.writeFileSync(configPath, JSON.stringify(config, null, 2));
266
283
  return {
@@ -271,4 +288,4 @@ function generateRun(scenarioDir, opts, paths) {
271
288
  };
272
289
  }
273
290
  //#endregion
274
- export { buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag };
291
+ export { DEFAULT_CLAUDE_AGENT, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
package/docs/adoption.md CHANGED
@@ -66,9 +66,11 @@ Commit `.skillcheck/scorecards/`. Gitignore the rest:
66
66
  .skillcheck/scratch/
67
67
  ```
68
68
 
69
- A scorecard is only comparable against the tree it graded, which is why every
70
- result carries the root repo's HEAD and `summarize` refuses to mix revisions
71
- without `--allow-mixed`.
69
+ A scorecard is only comparable against the tree it graded and the configuration
70
+ it ran with, which is why every result carries the root repo's HEAD, the agent
71
+ and judge models and efforts, and the trial count, and why `summarize` refuses
72
+ to mix them without `--allow-mixed`. Use `--trials 3` or more before calling a
73
+ skill change better or worse; one trial cannot tell a regression from noise.
72
74
 
73
75
  ## Upgrading
74
76
 
package/docs/scenarios.md CHANGED
@@ -54,7 +54,9 @@ item needs a non-empty `name` and `description` and a positive `max_score`.
54
54
  Each item becomes one `llm-rubric` assertion weighted by `max_score`, inside an
55
55
  assert-set with threshold 0.7. A separate `skill-used` assertion sits outside
56
56
  that aggregate, so a run that produces good output without ever loading the
57
- skill still fails. There is no test-level threshold: both must pass.
57
+ skill still fails. There is no test-level threshold: both must pass. The
58
+ reported score is the assert-set's weighted score; skill-used is reported
59
+ separately as a rate across trials.
58
60
 
59
61
  Write descriptions a judge can check against the deliverable: an observable
60
62
  property, not a feeling. Weight the items that would make a reviewer reject the
package/docs/usage.md CHANGED
@@ -35,9 +35,10 @@ skillcheck run skills/<skill>/evals/<scenario>
35
35
  skillcheck run <scenario-dir> --agent MODEL --judge MODEL --harness codex
36
36
  skillcheck run <scenario-dir> --harness claude --max-turns 80
37
37
  skillcheck run <scenario-dir> --harness grok
38
+ skillcheck run <scenario-dir> --trials 3 --agent-effort medium
38
39
  ```
39
40
 
40
- Materializes the scenario into `<root>/.skillcheck/scratch/<name>/workdir`,
41
+ Materializes the scenario into `<root>/.skillcheck/scratch/<name>/trial-<n>/workdir`,
41
42
  installs the skill under test into that workdir, drives the agent, and grades
42
43
  the files it wrote. Exit 0 means pass, 1 means graded fail, and 2 means error.
43
44
  Exit 2 covers missing usable promptfoo output or optional eval peers. The
@@ -53,6 +54,44 @@ and a Claude agent limit of 50 turns. `--max-turns` changes that limit only for
53
54
  Claude; passing it with `codex` or `grok` fails before the eval starts. On
54
55
  those harnesses, omitting `--agent` leaves the model to that CLI's own default.
55
56
 
57
+ ### Trials
58
+
59
+ One trial is one sample of a noisy process: the same scenario and skill can
60
+ score 0.49 and then 0.99. `--trials <k>` (default 1) runs the agent k times,
61
+ each in its own workdir with its own manifest, all graded in one promptfoo eval.
62
+ promptfoo's `--repeat` is not used because it reuses one set of vars and one
63
+ `working_dir`, so concurrent trials would write into the same tree and each
64
+ would be graded on all of their deliverables.
65
+
66
+ A scenario's result aggregates its trials:
67
+
68
+ | Field | Meaning |
69
+ | ------------------------------- | ------------------------------------------------------------------- |
70
+ | `pass` | pass^k: every trial passed. Exit 0 needs this |
71
+ | `passes`, `pass_rate` | Trials that passed, as a count and a fraction |
72
+ | `score` | Mean weighted checklist score (the assert-set, without skill-used) |
73
+ | `score_min` | Lowest trial score |
74
+ | `score_spread` | Highest minus lowest trial score |
75
+ | `skill_used`, `skill_used_rate` | Trials whose `skill-used` assertion passed, as a count and fraction |
76
+ | `noisy` | Trials both passed and failed, or `score_spread` is at least 0.2 |
77
+
78
+ A trial that errored was never graded, so one errored trial makes the whole
79
+ scenario an ERROR: pass^k over fewer than k trials is not the requested number.
80
+ So is a result with fewer rows than trials, or a row without its checklist and
81
+ `skill-used` components.
82
+ With `--trials` above 1, `run` and `sweep` print min, spread, pass count,
83
+ skill-used count, and `NOISY`.
84
+
85
+ ### Agent effort
86
+
87
+ `--agent-effort low|medium|high|xhigh|max` sets the Claude agent's effort.
88
+ promptfoo passes it to the Agent SDK, which starts Claude Code with `--effort`.
89
+ Omitting it leaves Claude Code's default. Only `--harness claude` takes it;
90
+ `codex` and `grok` fail before the eval starts. It is separate from
91
+ `--judge-effort`.
92
+
93
+ ### Harnesses and judges
94
+
56
95
  `--harness grok` runs the locally installed Grok Build CLI in the disposable
57
96
  workdir with the skill under `.grok/skills/`. It uses native streaming events
58
97
  to count a completed read of that skill's `SKILL.md` as `skill-used` evidence.
@@ -70,9 +109,16 @@ skillcheck run <scenario-dir> --judge openai:chat:gpt-5.6-sol --judge-effort hig
70
109
 
71
110
  A provider-qualified judge authenticates through that provider's own env
72
111
  (`OPENAI_API_KEY`, plus `OPENAI_BASE_URL` for a gateway) and is recorded
73
- verbatim in the scorecard's `judge_model` column. `--judge-effort`
74
- (minimal|low|medium|high) sets `reasoning_effort` and requires a
75
- provider-qualified judge; the Anthropic judge does not take one.
112
+ verbatim in the scorecard's `judge_model` column. For it, `--judge-effort`
113
+ (minimal|low|medium|high) sets `reasoning_effort`. For a bare Claude judge,
114
+ `--judge-effort` takes Claude's levels (low|medium|high|xhigh|max) and is passed
115
+ as `effort` on either Anthropic path; the SDK judge starts Claude Code with
116
+ `--effort`:
117
+
118
+ ```sh
119
+ skillcheck run <scenario-dir> --agent claude-opus-5-5 --agent-effort medium \
120
+ --judge claude-opus-5-5 --judge-effort high --trials 3
121
+ ```
76
122
 
77
123
  ## Sweep
78
124
 
@@ -86,7 +132,14 @@ order, sequentially. A scenario needs both `task.md` and `criteria.json` to be
86
132
  discovered. Exit 2 if anything errored, 1 if anything failed, else 0.
87
133
 
88
134
  `EVALS_CONCURRENCY` is passed to promptfoo as `-j` (default 4). It parallelizes
89
- within one scenario, not across them.
135
+ the trials of one scenario, not separate scenarios. To spread scenarios, start
136
+ several `skillcheck run` processes; each scenario has its own result, attempt
137
+ marker, and scratch directory.
138
+
139
+ A scenario is skipped only when its completed result was graded with the same
140
+ run configuration: agent model, agent effort, judge model, judge effort, and
141
+ trial count. A result from another configuration is rerun and reported as
142
+ `RERUN`.
90
143
 
91
144
  One known failure mode: judge calls through a gateway can drop at the transport
92
145
  layer ([uinaf/zebroid-infra#44](https://github.com/uinaf/zebroid-infra/issues/44)).
@@ -102,7 +155,10 @@ skillcheck summarize [--allow-mixed]
102
155
 
103
156
  Reduces `<root>/.skillcheck/results/*.json` into
104
157
  `<root>/.skillcheck/scorecards/<UTC-date>.json`: one entry per scenario with
105
- skill, scenario, harness, tree sha, score, pass, both models, latency, tokens.
158
+ skill, scenario, harness, tree sha, the trial aggregate from [trials](#trials),
159
+ the run configuration, mean latency per trial, and tokens summed over trials.
160
+ It then prints one row per skill and harness: scenarios, pass^k count, mean pass
161
+ rate, mean score, and noisy count. Each noisy scenario follows on its own line.
106
162
 
107
163
  If a scorecard for today already exists, the two are merged on
108
164
  `(skill, scenario, harness)`: entries from this run win, entries it did not
@@ -139,13 +195,23 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
139
195
  "skill": "<skill directory name>",
140
196
  "scenario": "<scenario directory name>",
141
197
  "harness": "claude",
198
+ "agent_model": "claude-opus-5",
199
+ "agent_effort": "medium",
200
+ "judge_model": "claude-opus-5",
201
+ "judge_effort": null,
202
+ "trials": 3,
203
+ "aggregate": { "pass": false, "pass_rate": 0.6667, "score": 0.81, "...": "..." },
142
204
  "ran_at": "<ISO timestamp>",
143
205
  "tool_version": "<skillcheck version>"
144
206
  }
145
207
  ```
146
208
 
147
- `summarize` reads those sidecars and refuses to mix skills-tree revisions in one
148
- scorecard, including retained rows from partial reruns, unless `--allow-mixed`.
209
+ `summarize` reads those sidecars and refuses to mix skills-tree revisions or
210
+ run configurations (agent model and effort, judge model and effort, trials) in
211
+ one scorecard, including retained rows from partial reruns, unless
212
+ `--allow-mixed`. Configurations are compared within a harness, since harnesses
213
+ differ by design. Sidecars written before run configurations were recorded fall
214
+ back to the promptfoo config stored in the result.
149
215
  Rejection leaves the existing scorecard unchanged. With the override, the top-level `skills_tree_sha`
150
216
  becomes `mixed` and per-entry shas remain. A result with no sidecar reduces as
151
217
  `unattested`.
@@ -169,3 +235,10 @@ written inside the installed package.
169
235
 
170
236
  A bare `--judge` model stays on the Anthropic selection regardless of the
171
237
  agent harness; a provider-qualified `--judge` uses that provider's env instead.
238
+
239
+ The Claude agent loads project settings only, so an `apiKeyHelper` or `env`
240
+ block in the operator's `~/.claude/settings.json` never reaches it; the run
241
+ fails with `Not logged in`. Export the gateway variables instead:
242
+ `ANTHROPIC_BASE_URL` and `ANTHROPIC_AUTH_TOKEN` set to the helper's output.
243
+ Running from inside a Claude Code session also leaks that session's
244
+ `CLAUDECODE` and `CLAUDE_CODE_*` variables into the agent; run from a plain shell.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "1.0.2",
3
+ "version": "1.2.0",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {
@@ -33,17 +33,17 @@
33
33
  "prepublishOnly": "pnpm run verify:full"
34
34
  },
35
35
  "devDependencies": {
36
- "@anthropic-ai/claude-agent-sdk": "^0.3.233",
37
- "@openai/codex-sdk": "^0.147.0",
38
- "@types/node": "^26.2.0",
39
- "promptfoo": "^0.122.0",
36
+ "@anthropic-ai/claude-agent-sdk": "^0.3.281",
37
+ "@openai/codex-sdk": "^0.156.1",
38
+ "@types/node": "^26.6.2",
39
+ "promptfoo": "^0.123.1",
40
40
  "vite": "catalog:",
41
41
  "vite-plus": "catalog:"
42
42
  },
43
43
  "peerDependencies": {
44
- "@anthropic-ai/claude-agent-sdk": "^0.3.233",
45
- "@openai/codex-sdk": "^0.147.0",
46
- "promptfoo": "^0.122.0"
44
+ "@anthropic-ai/claude-agent-sdk": "^0.3.281",
45
+ "@openai/codex-sdk": "^0.156.1",
46
+ "promptfoo": "^0.123.1"
47
47
  },
48
48
  "peerDependenciesMeta": {
49
49
  "@anthropic-ai/claude-agent-sdk": {
@@ -56,8 +56,15 @@
56
56
  "optional": true
57
57
  }
58
58
  },
59
+ "devEngines": {
60
+ "runtime": {
61
+ "name": "node",
62
+ "version": "^24.11.0 || >=26.0.0",
63
+ "onFail": "error"
64
+ }
65
+ },
59
66
  "engines": {
60
67
  "node": ">=24"
61
68
  },
62
- "packageManager": "pnpm@12.0.0"
69
+ "packageManager": "pnpm@12.4.2"
63
70
  }