@uinaf/skillcheck 1.0.1 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -1
- package/dist/cli.js +281 -75
- package/dist/scenario.js +32 -11
- package/docs/adoption.md +5 -3
- package/docs/scenarios.md +3 -1
- package/docs/usage.md +80 -7
- package/package.json +16 -9
package/README.md
CHANGED
|
@@ -28,7 +28,8 @@ pre-npm git tags are covered there too.
|
|
|
28
28
|
```sh
|
|
29
29
|
skillcheck lint # structural lint, no credentials
|
|
30
30
|
skillcheck run skills/wat/evals/basic # one scenario, graded end to end
|
|
31
|
-
skillcheck sweep
|
|
31
|
+
skillcheck sweep --trials 3 # every scenario, three trials each
|
|
32
|
+
skillcheck summarize # per-scenario and per-skill scorecard
|
|
32
33
|
```
|
|
33
34
|
|
|
34
35
|
`lint` needs nothing. `run` and `sweep` need model auth, which is why sweeps
|
package/dist/cli.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import { lintSkills } from "./lint.js";
|
|
3
|
-
import { generateRun, requiredEvalPackages, resolvePackageDir, runNameFor } from "./scenario.js";
|
|
3
|
+
import { encodeRunNamePart, generateRun, requiredEvalPackages, resolvePackageDir, runNameFor } from "./scenario.js";
|
|
4
4
|
import { execFileSync, spawnSync } from "node:child_process";
|
|
5
5
|
import fs from "node:fs";
|
|
6
6
|
import path from "node:path";
|
|
@@ -12,11 +12,18 @@ const selfExt = path.extname(fileURLToPath(import.meta.url));
|
|
|
12
12
|
function toolVersion() {
|
|
13
13
|
return JSON.parse(fs.readFileSync(path.join(packageDir, "package.json"), "utf8")).version;
|
|
14
14
|
}
|
|
15
|
-
function
|
|
15
|
+
function parsePositiveInt(flag, raw) {
|
|
16
16
|
const n = Number(raw);
|
|
17
|
-
if (!Number.isInteger(n) || n <= 0) throw new Error(
|
|
17
|
+
if (!Number.isInteger(n) || n <= 0) throw new Error(`${flag} must be a positive integer, got ${JSON.stringify(raw)}`);
|
|
18
18
|
return n;
|
|
19
19
|
}
|
|
20
|
+
const AGENT_EFFORTS = [
|
|
21
|
+
"low",
|
|
22
|
+
"medium",
|
|
23
|
+
"high",
|
|
24
|
+
"xhigh",
|
|
25
|
+
"max"
|
|
26
|
+
];
|
|
20
27
|
function fail(msg) {
|
|
21
28
|
console.error(msg);
|
|
22
29
|
process.exit(1);
|
|
@@ -29,8 +36,10 @@ function parseArgs(argv) {
|
|
|
29
36
|
"--agent",
|
|
30
37
|
"--judge",
|
|
31
38
|
"--judge-effort",
|
|
39
|
+
"--agent-effort",
|
|
32
40
|
"--harness",
|
|
33
|
-
"--max-turns"
|
|
41
|
+
"--max-turns",
|
|
42
|
+
"--trials"
|
|
34
43
|
]);
|
|
35
44
|
for (let i = 0; i < argv.length; i++) {
|
|
36
45
|
const a = argv[i];
|
|
@@ -60,8 +69,13 @@ function stateDirs(root) {
|
|
|
60
69
|
}
|
|
61
70
|
function runOptions(flags) {
|
|
62
71
|
const harness = flags.get("--harness") ?? "claude";
|
|
63
|
-
if (harness !== "claude" && harness !== "codex" && harness !== "grok")
|
|
64
|
-
if (flags.has("--max-turns") && harness !== "claude")
|
|
72
|
+
if (harness !== "claude" && harness !== "codex" && harness !== "grok") throw new Error(`--harness must be claude, codex, or grok, got ${harness}`);
|
|
73
|
+
if (flags.has("--max-turns") && harness !== "claude") throw new Error("--max-turns is only supported with --harness claude");
|
|
74
|
+
const agentEffort = flags.get("--agent-effort");
|
|
75
|
+
if (agentEffort !== void 0) {
|
|
76
|
+
if (harness !== "claude") throw new Error(`--agent-effort is only supported with --harness claude; ${harness} has no effort setting wired`);
|
|
77
|
+
if (!AGENT_EFFORTS.includes(agentEffort)) throw new Error(`--agent-effort must be ${AGENT_EFFORTS.join(", ")}, got ${agentEffort}`);
|
|
78
|
+
}
|
|
65
79
|
const agent = flags.get("--agent");
|
|
66
80
|
const judgeModel = flags.get("--judge") ?? "claude-opus-5";
|
|
67
81
|
const judgeEffort = flags.get("--judge-effort");
|
|
@@ -71,17 +85,51 @@ function runOptions(flags) {
|
|
|
71
85
|
"low",
|
|
72
86
|
"medium",
|
|
73
87
|
"high"
|
|
74
|
-
].includes(judgeEffort))
|
|
75
|
-
if (!judgeModel.includes(":"))
|
|
88
|
+
].includes(judgeEffort)) throw new Error(`--judge-effort must be minimal, low, medium, or high, got ${judgeEffort}`);
|
|
89
|
+
if (!judgeModel.includes(":")) throw new Error("--judge-effort needs a provider-qualified --judge (e.g. openai:chat:gpt-5.6-sol); the Anthropic judge does not take a reasoning effort");
|
|
76
90
|
}
|
|
77
91
|
return {
|
|
78
92
|
harness,
|
|
79
93
|
agentModel: agent,
|
|
94
|
+
agentEffort,
|
|
80
95
|
judgeModel,
|
|
81
96
|
judgeEffort,
|
|
82
|
-
maxTurns: flags.has("--max-turns") ?
|
|
97
|
+
maxTurns: flags.has("--max-turns") ? parsePositiveInt("--max-turns", flags.get("--max-turns")) : void 0,
|
|
98
|
+
trials: flags.has("--trials") ? parsePositiveInt("--trials", flags.get("--trials")) : 1
|
|
99
|
+
};
|
|
100
|
+
}
|
|
101
|
+
function runConfigOf(opts) {
|
|
102
|
+
return {
|
|
103
|
+
agent_model: opts.agentModel ?? (opts.harness === "claude" ? "claude-opus-5" : `${opts.harness}-default`),
|
|
104
|
+
agent_effort: opts.agentEffort ?? null,
|
|
105
|
+
judge_model: opts.judgeModel,
|
|
106
|
+
judge_effort: opts.judgeEffort ?? null,
|
|
107
|
+
trials: opts.trials ?? 1
|
|
83
108
|
};
|
|
84
109
|
}
|
|
110
|
+
function configKey(c) {
|
|
111
|
+
return JSON.stringify([
|
|
112
|
+
c.agent_model,
|
|
113
|
+
c.agent_effort ?? null,
|
|
114
|
+
c.judge_model,
|
|
115
|
+
c.judge_effort ?? null,
|
|
116
|
+
c.trials ?? 1
|
|
117
|
+
]);
|
|
118
|
+
}
|
|
119
|
+
function describeConfig(c) {
|
|
120
|
+
const effort = (e) => e ? `@${e}` : "";
|
|
121
|
+
return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}`;
|
|
122
|
+
}
|
|
123
|
+
function assertUniformConfig(entries, allowMixed) {
|
|
124
|
+
if (allowMixed) return;
|
|
125
|
+
const byHarness = /* @__PURE__ */ new Map();
|
|
126
|
+
for (const e of entries) {
|
|
127
|
+
const configs = byHarness.get(e.harness) ?? /* @__PURE__ */ new Map();
|
|
128
|
+
configs.set(configKey(e), e);
|
|
129
|
+
byHarness.set(e.harness, configs);
|
|
130
|
+
}
|
|
131
|
+
for (const [harness, configs] of byHarness) if (configs.size > 1) throw new Error(`${harness} results span multiple run configurations (${[...configs.values()].map(describeConfig).join("; ")}); rerun them to match or pass --allow-mixed`);
|
|
132
|
+
}
|
|
85
133
|
function ensureEvalPackages(opts) {
|
|
86
134
|
const missing = requiredEvalPackages(opts, process.env.ANTHROPIC_API_KEY !== void 0).filter((pkg) => resolvePackageDir(pkg) === void 0);
|
|
87
135
|
if (missing.length === 0) return;
|
|
@@ -105,21 +153,76 @@ function promptfooEntry() {
|
|
|
105
153
|
if (typeof rel !== "string") throw new Error(`promptfoo at ${dir} declares no bin`);
|
|
106
154
|
return path.join(dir, rel);
|
|
107
155
|
}
|
|
108
|
-
|
|
156
|
+
const GRADED_REASONS = /* @__PURE__ */ new Set([0, 1]);
|
|
157
|
+
const ERRORED_REASON = 2;
|
|
158
|
+
function classifyResult(raw, expectedTrials) {
|
|
109
159
|
const root = raw;
|
|
110
|
-
const
|
|
160
|
+
const rows = root?.results?.results;
|
|
161
|
+
if (!Array.isArray(rows) || rows.length === 0) return { error: "promptfoo output carried no result" };
|
|
162
|
+
if (expectedTrials !== void 0 && rows.length !== expectedTrials) return { error: `promptfoo returned ${rows.length} of ${expectedTrials} trials` };
|
|
163
|
+
const stats = rows.length === 1 ? root?.results?.stats ?? void 0 : void 0;
|
|
164
|
+
const trials = [];
|
|
165
|
+
for (const [i, row] of rows.entries()) {
|
|
166
|
+
const verdict = classifyRow(row, stats);
|
|
167
|
+
if ("error" in verdict) return { error: rows.length === 1 ? verdict.error : `trial ${i + 1}: ${verdict.error}` };
|
|
168
|
+
trials.push(verdict);
|
|
169
|
+
}
|
|
170
|
+
return { trials };
|
|
171
|
+
}
|
|
172
|
+
function classifyRow(raw, stats) {
|
|
173
|
+
const res = raw;
|
|
111
174
|
if (res === void 0 || res === null) return { error: "promptfoo output carried no result" };
|
|
112
175
|
const message = typeof res.error === "string" ? res.error.trim() : "";
|
|
113
|
-
|
|
114
|
-
const
|
|
115
|
-
if (message !== "" && !
|
|
116
|
-
if (stats !== void 0 && (stats.errors ?? 0) > 0 && (stats.successes ?? 0) === 0 && (stats.failures ?? 0) === 0) return { error: "promptfoo reported an errored test with nothing graded" };
|
|
176
|
+
if (res.failureReason === ERRORED_REASON) return { error: message || "promptfoo reported an errored test" };
|
|
177
|
+
const graded = GRADED_REASONS.has(res.failureReason) || stats !== void 0 && (stats.errors ?? 0) === 0 && ((stats.failures ?? 0) > 0 || (stats.successes ?? 0) > 0);
|
|
178
|
+
if (message !== "" && !graded) return { error: message };
|
|
179
|
+
if (!graded && stats !== void 0 && (stats.errors ?? 0) > 0 && (stats.successes ?? 0) === 0 && (stats.failures ?? 0) === 0) return { error: "promptfoo reported an errored test with nothing graded" };
|
|
117
180
|
if (typeof res.score !== "number" || typeof res.success !== "boolean") return { error: "promptfoo result carried no usable score" };
|
|
181
|
+
const components = Array.isArray(res.gradingResult?.componentResults) ? res.gradingResult.componentResults : [];
|
|
182
|
+
const checklist = components.find((c) => c?.metadata?.assertionSet?.type === "assert-set");
|
|
183
|
+
const skillUsed = components.find((c) => c?.assertion?.type === "skill-used");
|
|
184
|
+
if (typeof checklist?.score !== "number" || typeof skillUsed?.pass !== "boolean") return { error: "promptfoo result carried no checklist or skill-used verdict" };
|
|
118
185
|
return {
|
|
119
|
-
score:
|
|
120
|
-
pass: res.success
|
|
186
|
+
score: checklist.score,
|
|
187
|
+
pass: res.success,
|
|
188
|
+
skillUsed: skillUsed.pass
|
|
121
189
|
};
|
|
122
190
|
}
|
|
191
|
+
const NOISY_SPREAD = .2;
|
|
192
|
+
const round4 = (n) => Math.round(n * 1e4) / 1e4;
|
|
193
|
+
function aggregateTrials(trials) {
|
|
194
|
+
if (trials.length === 0) throw new Error("cannot aggregate zero trials");
|
|
195
|
+
const scores = trials.map((t) => t.score);
|
|
196
|
+
const passes = trials.filter((t) => t.pass).length;
|
|
197
|
+
const min = Math.min(...scores);
|
|
198
|
+
const spread = Math.max(...scores) - min;
|
|
199
|
+
const skillUsed = trials.filter((t) => t.skillUsed).length;
|
|
200
|
+
return {
|
|
201
|
+
trials: trials.length,
|
|
202
|
+
pass: passes === trials.length,
|
|
203
|
+
passes,
|
|
204
|
+
pass_rate: round4(passes / trials.length),
|
|
205
|
+
score: round4(scores.reduce((a, b) => a + b, 0) / trials.length),
|
|
206
|
+
score_min: round4(min),
|
|
207
|
+
score_spread: round4(spread),
|
|
208
|
+
skill_used: skillUsed,
|
|
209
|
+
skill_used_rate: round4(skillUsed / trials.length),
|
|
210
|
+
noisy: passes > 0 && passes < trials.length || spread >= .199999999
|
|
211
|
+
};
|
|
212
|
+
}
|
|
213
|
+
function formatStats(s) {
|
|
214
|
+
const line = `score=${s.score.toFixed(4)}`;
|
|
215
|
+
if (s.trials === 1) return line;
|
|
216
|
+
return [
|
|
217
|
+
line,
|
|
218
|
+
`min=${s.score_min.toFixed(4)}`,
|
|
219
|
+
`spread=${s.score_spread.toFixed(4)}`,
|
|
220
|
+
`pass^${s.trials}=${s.pass ? "yes" : "no"}`,
|
|
221
|
+
`passes=${s.passes}/${s.trials}`,
|
|
222
|
+
`skill-used=${s.skill_used}/${s.trials}`,
|
|
223
|
+
...s.noisy ? ["NOISY"] : []
|
|
224
|
+
].join(" ");
|
|
225
|
+
}
|
|
123
226
|
function gitHead(root) {
|
|
124
227
|
return execFileSync("git", ["rev-parse", "HEAD"], {
|
|
125
228
|
cwd: root,
|
|
@@ -134,14 +237,19 @@ function attemptPath(resultPath) {
|
|
|
134
237
|
}
|
|
135
238
|
function runScenario(scenarioDir, opts, root) {
|
|
136
239
|
const dirs = stateDirs(root);
|
|
137
|
-
const { name, configPath } = generateRun(path.resolve(scenarioDir), opts, {
|
|
240
|
+
const { name, configPath, skill, scenario } = generateRun(path.resolve(scenarioDir), opts, {
|
|
138
241
|
scratchDir: dirs.scratch,
|
|
139
242
|
transformPath: path.join(here, `transform${selfExt}`),
|
|
140
243
|
grokProviderPath: path.join(here, `grok-provider${selfExt}`)
|
|
141
244
|
});
|
|
142
245
|
fs.mkdirSync(dirs.results, { recursive: true });
|
|
143
246
|
const resultPath = path.join(dirs.results, `${name}.json`);
|
|
144
|
-
|
|
247
|
+
const identity = {
|
|
248
|
+
skill,
|
|
249
|
+
scenario,
|
|
250
|
+
harness: opts.harness
|
|
251
|
+
};
|
|
252
|
+
fs.writeFileSync(attemptPath(resultPath), JSON.stringify(identity) + "\n");
|
|
145
253
|
fs.rmSync(resultPath, { force: true });
|
|
146
254
|
fs.rmSync(metaPath(resultPath), { force: true });
|
|
147
255
|
const sha = gitHead(root);
|
|
@@ -172,24 +280,60 @@ function runScenario(scenarioDir, opts, root) {
|
|
|
172
280
|
if (rc !== 0) return outcome;
|
|
173
281
|
let verdict;
|
|
174
282
|
try {
|
|
175
|
-
verdict = classifyResult(JSON.parse(fs.readFileSync(resultPath, "utf8")));
|
|
283
|
+
verdict = classifyResult(JSON.parse(fs.readFileSync(resultPath, "utf8")), opts.trials ?? 1);
|
|
176
284
|
} catch {
|
|
177
285
|
verdict = { error: "promptfoo produced no parseable result file" };
|
|
178
286
|
}
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
287
|
+
if ("error" in verdict) return {
|
|
288
|
+
...outcome,
|
|
289
|
+
error: verdict.error
|
|
290
|
+
};
|
|
291
|
+
outcome.stats = aggregateTrials(verdict.trials);
|
|
292
|
+
fs.writeFileSync(metaPath(resultPath), JSON.stringify({
|
|
293
|
+
skills_tree_sha: sha,
|
|
294
|
+
...identity,
|
|
295
|
+
...runConfigOf(opts),
|
|
296
|
+
aggregate: outcome.stats,
|
|
297
|
+
ran_at: (/* @__PURE__ */ new Date()).toISOString(),
|
|
298
|
+
tool_version: toolVersion()
|
|
299
|
+
}, null, 2) + "\n");
|
|
300
|
+
fs.rmSync(attemptPath(resultPath), { force: true });
|
|
191
301
|
return outcome;
|
|
192
302
|
}
|
|
303
|
+
function judgeName(judge) {
|
|
304
|
+
if (typeof judge === "string") return judge.replace(/^anthropic:messages:/, "");
|
|
305
|
+
const j = judge;
|
|
306
|
+
const name = j?.config?.model ?? j?.id;
|
|
307
|
+
return typeof name === "string" ? name : "unknown";
|
|
308
|
+
}
|
|
309
|
+
function resultRunConfig(raw, meta, harness) {
|
|
310
|
+
const m = meta;
|
|
311
|
+
if (typeof m?.agent_model === "string" && typeof m.judge_model === "string") return {
|
|
312
|
+
agent_model: m.agent_model,
|
|
313
|
+
agent_effort: m.agent_effort ?? null,
|
|
314
|
+
judge_model: m.judge_model,
|
|
315
|
+
judge_effort: m.judge_effort ?? null,
|
|
316
|
+
trials: m.trials ?? 1
|
|
317
|
+
};
|
|
318
|
+
const r = raw;
|
|
319
|
+
const agent = r?.config?.providers?.[0]?.config;
|
|
320
|
+
const judge = r?.config?.defaultTest?.options?.provider;
|
|
321
|
+
const judgeEffort = judge?.config?.reasoning_effort;
|
|
322
|
+
return {
|
|
323
|
+
agent_model: typeof agent?.model === "string" ? agent.model : `${harness}-default`,
|
|
324
|
+
agent_effort: typeof agent?.effort === "string" ? agent.effort : null,
|
|
325
|
+
judge_model: judgeName(judge),
|
|
326
|
+
judge_effort: typeof judgeEffort === "string" ? judgeEffort : null,
|
|
327
|
+
trials: r?.results?.results?.length ?? 1
|
|
328
|
+
};
|
|
329
|
+
}
|
|
330
|
+
function readJson(file) {
|
|
331
|
+
try {
|
|
332
|
+
return JSON.parse(fs.readFileSync(file, "utf8"));
|
|
333
|
+
} catch {
|
|
334
|
+
return;
|
|
335
|
+
}
|
|
336
|
+
}
|
|
193
337
|
function discoverScenarios(root) {
|
|
194
338
|
const roots = [path.join(root, "skills")];
|
|
195
339
|
const cliDir = path.join(root, "cli");
|
|
@@ -210,61 +354,79 @@ function discoverScenarios(root) {
|
|
|
210
354
|
}
|
|
211
355
|
return found.sort();
|
|
212
356
|
}
|
|
357
|
+
const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N]";
|
|
213
358
|
function cmdRun(argv) {
|
|
214
359
|
const { positional, flags } = parseArgs(argv);
|
|
215
|
-
if (positional.length !== 1) fail(
|
|
360
|
+
if (positional.length !== 1) fail(`usage: skillcheck run <scenario-dir> ${RUN_FLAGS}`);
|
|
216
361
|
const opts = runOptions(flags);
|
|
217
362
|
ensureEvalPackages(opts);
|
|
218
363
|
const o = runScenario(positional[0], opts, resolveRoot(flags));
|
|
219
|
-
if (o.
|
|
364
|
+
if (o.stats === void 0) {
|
|
220
365
|
console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
|
|
221
366
|
process.exit(2);
|
|
222
367
|
}
|
|
223
|
-
console.log(`${o.pass ? "PASS" : "FAIL"} ${o.name}
|
|
224
|
-
process.exit(o.pass ? 0 : 1);
|
|
368
|
+
console.log(`${o.stats.pass ? "PASS" : "FAIL"} ${o.name} ${formatStats(o.stats)} (results: ${o.resultPath})`);
|
|
369
|
+
process.exit(o.stats.pass ? 0 : 1);
|
|
225
370
|
}
|
|
226
371
|
function cmdSweep(argv) {
|
|
227
372
|
const { positional, flags } = parseArgs(argv);
|
|
228
|
-
if (positional.length > 0) fail(
|
|
373
|
+
if (positional.length > 0) fail(`usage: skillcheck sweep ${RUN_FLAGS} [--all]`);
|
|
229
374
|
const root = resolveRoot(flags);
|
|
230
375
|
const opts = runOptions(flags);
|
|
231
376
|
ensureEvalPackages(opts);
|
|
232
377
|
const all = flags.get("--all") === true;
|
|
233
378
|
const resultsDir = stateDirs(root).results;
|
|
379
|
+
const wanted = configKey(runConfigOf(opts));
|
|
234
380
|
let passed = 0, failed = 0, errored = 0, skipped = 0;
|
|
235
381
|
for (const dir of discoverScenarios(root)) {
|
|
236
382
|
const name = runNameFor(dir, opts.harness);
|
|
237
383
|
const resultPath = path.join(resultsDir, `${name}.json`);
|
|
238
384
|
if (!all && fs.existsSync(resultPath) && !fs.existsSync(attemptPath(resultPath))) {
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
385
|
+
const raw = readJson(resultPath);
|
|
386
|
+
const config = resultRunConfig(raw, readJson(metaPath(resultPath)), opts.harness);
|
|
387
|
+
const graded = !("error" in classifyResult(raw, config.trials));
|
|
388
|
+
if (graded && configKey(config) === wanted) {
|
|
389
|
+
skipped++;
|
|
390
|
+
console.log(`SKIP ${name} (results exist; use --all to rerun)`);
|
|
391
|
+
continue;
|
|
392
|
+
}
|
|
393
|
+
if (graded) console.log(`RERUN ${name} (results used ${describeConfig(config)})`);
|
|
242
394
|
}
|
|
243
395
|
const o = runScenario(dir, opts, root);
|
|
244
|
-
if (o.
|
|
396
|
+
if (o.stats === void 0) {
|
|
245
397
|
errored++;
|
|
246
398
|
console.log(`ERROR ${o.name} ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
|
|
247
|
-
} else if (o.pass) {
|
|
399
|
+
} else if (o.stats.pass) {
|
|
248
400
|
passed++;
|
|
249
|
-
console.log(`PASS ${o.name}
|
|
401
|
+
console.log(`PASS ${o.name} ${formatStats(o.stats)}`);
|
|
250
402
|
} else {
|
|
251
403
|
failed++;
|
|
252
|
-
console.log(`FAIL ${o.name}
|
|
404
|
+
console.log(`FAIL ${o.name} ${formatStats(o.stats)}`);
|
|
253
405
|
}
|
|
254
406
|
}
|
|
255
407
|
console.log(`\nsweep: ${passed} passed, ${failed} failed, ${errored} errored, ${skipped} skipped`);
|
|
256
408
|
process.exit(errored > 0 ? 2 : failed > 0 ? 1 : 0);
|
|
257
409
|
}
|
|
258
|
-
function resultIdentity(file) {
|
|
410
|
+
function resultIdentity(file, dir) {
|
|
411
|
+
const base = file.replace(/\.json$/, "");
|
|
412
|
+
for (const sidecar of [`${base}.meta.json`, `${file}.attempt`]) try {
|
|
413
|
+
const identity = JSON.parse(fs.readFileSync(path.join(dir, sidecar), "utf8"));
|
|
414
|
+
if (identity !== null && typeof identity === "object" && "skill" in identity && typeof identity.skill === "string" && "scenario" in identity && typeof identity.scenario === "string" && "harness" in identity && (identity.harness === "claude" || identity.harness === "codex" || identity.harness === "grok")) return {
|
|
415
|
+
skill: identity.skill,
|
|
416
|
+
scenario: identity.scenario,
|
|
417
|
+
harness: identity.harness
|
|
418
|
+
};
|
|
419
|
+
} catch {}
|
|
420
|
+
if (/^~v3~[0-9a-f]{64}$/.test(base)) throw new Error(`missing identity metadata for ${file}`);
|
|
259
421
|
const decode = (part) => {
|
|
260
422
|
if (!part.startsWith("~v2~")) return part;
|
|
261
423
|
try {
|
|
262
|
-
|
|
424
|
+
const decoded = decodeURIComponent(part.slice(4));
|
|
425
|
+
return encodeRunNamePart(decoded) === part ? decoded : part;
|
|
263
426
|
} catch {
|
|
264
427
|
return part;
|
|
265
428
|
}
|
|
266
429
|
};
|
|
267
|
-
const base = file.replace(/\.json$/, "");
|
|
268
430
|
const suffix = base.match(/--(codex|grok|cursor)$/);
|
|
269
431
|
const harness = suffix === null ? "claude" : suffix[1];
|
|
270
432
|
const [skill, ...rest] = base.replace(/--(codex|grok|cursor)$/, "").split("--");
|
|
@@ -277,6 +439,7 @@ function resultIdentity(file) {
|
|
|
277
439
|
function reduceResults(dir, allowMixed) {
|
|
278
440
|
const entries = [];
|
|
279
441
|
const skipped = [];
|
|
442
|
+
const gradedAt = /* @__PURE__ */ new Map();
|
|
280
443
|
const shas = /* @__PURE__ */ new Set();
|
|
281
444
|
const files = fs.readdirSync(dir);
|
|
282
445
|
const incomplete = new Set(files.filter((f) => f.endsWith(".json.attempt")).map((f) => f.replace(/\.attempt$/, "")));
|
|
@@ -292,46 +455,44 @@ function reduceResults(dir, allowMixed) {
|
|
|
292
455
|
skipped.push(f);
|
|
293
456
|
continue;
|
|
294
457
|
}
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
}
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
const verdict = classifyResult(raw);
|
|
303
|
-
if (verdict.score === void 0 || verdict.pass === void 0) {
|
|
458
|
+
const raw = readJson(path.join(dir, f));
|
|
459
|
+
const base = f.replace(/\.json$/, "");
|
|
460
|
+
const meta = readJson(path.join(dir, `${base}.meta.json`));
|
|
461
|
+
const { skill, scenario, harness } = resultIdentity(f, dir);
|
|
462
|
+
const config = resultRunConfig(raw, meta, harness);
|
|
463
|
+
const verdict = classifyResult(raw, config.trials);
|
|
464
|
+
if ("error" in verdict) {
|
|
304
465
|
console.error(`skipping ${f}: ${verdict.error}`);
|
|
305
466
|
skipped.push(f);
|
|
306
467
|
continue;
|
|
307
468
|
}
|
|
308
|
-
const
|
|
309
|
-
const
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
469
|
+
const rows = raw.results.results;
|
|
470
|
+
const key = entryKey({
|
|
471
|
+
skill,
|
|
472
|
+
scenario,
|
|
473
|
+
harness
|
|
474
|
+
});
|
|
475
|
+
gradedAt.set(key, Math.max(gradedAt.get(key) ?? 0, fs.statSync(path.join(dir, f)).mtimeMs));
|
|
476
|
+
const sha = typeof meta?.skills_tree_sha === "string" ? meta.skills_tree_sha : "unattested";
|
|
316
477
|
shas.add(sha);
|
|
478
|
+
const stats = aggregateTrials(verdict.trials);
|
|
317
479
|
entries.push({
|
|
318
480
|
skill,
|
|
319
481
|
scenario,
|
|
320
482
|
harness,
|
|
321
483
|
skills_tree_sha: sha,
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
latency_ms: res.latencyMs,
|
|
327
|
-
tokens: (res.tokenUsage?.total ?? 0) + (res.tokenUsage?.assertions?.total ?? 0)
|
|
484
|
+
...stats,
|
|
485
|
+
...config,
|
|
486
|
+
latency_ms: Math.round(rows.reduce((a, r) => a + (r.latencyMs ?? 0), 0) / rows.length),
|
|
487
|
+
tokens: rows.reduce((a, r) => a + (r.tokenUsage?.total ?? 0) + (r.tokenUsage?.assertions?.total ?? 0), 0)
|
|
328
488
|
});
|
|
329
489
|
}
|
|
330
490
|
if (shas.size > 1 && !allowMixed) throw new Error(`results span multiple skills-tree revisions (${[...shas].join(", ")}); rerun stale ones or pass --allow-mixed`);
|
|
331
491
|
return {
|
|
332
492
|
treeSha: shas.size === 1 ? [...shas][0] : shas.size === 0 ? "none" : "mixed",
|
|
333
493
|
entries,
|
|
334
|
-
skipped
|
|
494
|
+
skipped,
|
|
495
|
+
gradedAt
|
|
335
496
|
};
|
|
336
497
|
}
|
|
337
498
|
function entryKey(e) {
|
|
@@ -375,15 +536,20 @@ function cmdSummarize(argv) {
|
|
|
375
536
|
if (positional.length > 0) fail("usage: skillcheck summarize [--root DIR] [--allow-mixed]");
|
|
376
537
|
const dirs = stateDirs(resolveRoot(flags));
|
|
377
538
|
if (!fs.existsSync(dirs.results)) fail(`no results directory at ${dirs.results}; run some evals first`);
|
|
378
|
-
const { entries, skipped } = reduceResults(dirs.results, flags.get("--allow-mixed") === true);
|
|
539
|
+
const { entries, skipped, gradedAt } = reduceResults(dirs.results, flags.get("--allow-mixed") === true);
|
|
379
540
|
fs.mkdirSync(dirs.scorecards, { recursive: true });
|
|
380
541
|
const out = path.join(dirs.scorecards, `${(/* @__PURE__ */ new Date()).toISOString().slice(0, 10)}.json`);
|
|
381
542
|
const existing = readExistingScorecard(out);
|
|
382
|
-
const skippedKeys = new Set(skipped.map((file) =>
|
|
543
|
+
const skippedKeys = new Set(skipped.map((file) => ({
|
|
544
|
+
key: entryKey(resultIdentity(file, dirs.results)),
|
|
545
|
+
modifiedAt: fs.statSync(fs.existsSync(path.join(dirs.results, `${file}.attempt`)) ? path.join(dirs.results, `${file}.attempt`) : path.join(dirs.results, file)).mtimeMs
|
|
546
|
+
})).filter(({ key, modifiedAt }) => (gradedAt.get(key) ?? 0) <= modifiedAt).map(({ key }) => key));
|
|
383
547
|
if (existing.some((entry) => skippedKeys.has(entryKey(entry)))) throw new Error("skipped rerun matches an existing score; refusing to carry it or overwrite the scorecard");
|
|
384
548
|
const merged = mergeScorecard(existing, entries);
|
|
385
549
|
const treeSha = treeShaOf(merged.entries);
|
|
386
|
-
|
|
550
|
+
const allowMixed = flags.get("--allow-mixed") === true;
|
|
551
|
+
if (treeSha === "mixed" && !allowMixed) throw new Error("scorecard spans multiple skills-tree revisions; rerun stale ones or pass --allow-mixed");
|
|
552
|
+
assertUniformConfig(merged.entries, allowMixed);
|
|
387
553
|
const scorecard = {
|
|
388
554
|
ran_at: (/* @__PURE__ */ new Date()).toISOString(),
|
|
389
555
|
skills_tree_sha: treeSha,
|
|
@@ -392,6 +558,46 @@ function cmdSummarize(argv) {
|
|
|
392
558
|
fs.writeFileSync(out, JSON.stringify(scorecard, null, 2) + "\n");
|
|
393
559
|
console.log(`${out}: ${merged.entries.length} scenario(s), ${merged.entries.filter((e) => e.pass).length} passing, ${skipped.length} skipped file(s)`);
|
|
394
560
|
if (existing.length > 0) console.log(`merged into today's scorecard: ${entries.length} from this run, ${merged.carried} carried over`);
|
|
561
|
+
console.log(`\n${formatSkillTable(summarizeSkills(merged.entries))}`);
|
|
562
|
+
for (const e of merged.entries.filter((x) => x.noisy)) console.log(`NOISY ${e.skill}/${e.scenario} (${e.harness}) ${formatStats(e)}`);
|
|
563
|
+
}
|
|
564
|
+
function summarizeSkills(entries) {
|
|
565
|
+
const groups = /* @__PURE__ */ new Map();
|
|
566
|
+
for (const e of entries) {
|
|
567
|
+
const key = `${e.skill}\0${e.harness}`;
|
|
568
|
+
groups.set(key, [...groups.get(key) ?? [], e]);
|
|
569
|
+
}
|
|
570
|
+
const mean = (xs) => round4(xs.reduce((a, b) => a + b, 0) / xs.length);
|
|
571
|
+
return [...groups.values()].map((rows) => ({
|
|
572
|
+
skill: rows[0].skill,
|
|
573
|
+
harness: rows[0].harness,
|
|
574
|
+
scenarios: rows.length,
|
|
575
|
+
pass_all: rows.filter((r) => r.pass).length,
|
|
576
|
+
pass_rate: mean(rows.map((r) => r.pass_rate ?? (r.pass ? 1 : 0))),
|
|
577
|
+
score: mean(rows.map((r) => r.score)),
|
|
578
|
+
noisy: rows.filter((r) => r.noisy).map((r) => r.scenario)
|
|
579
|
+
}));
|
|
580
|
+
}
|
|
581
|
+
function formatSkillTable(rows) {
|
|
582
|
+
const table = [[
|
|
583
|
+
"skill",
|
|
584
|
+
"harness",
|
|
585
|
+
"scenarios",
|
|
586
|
+
"pass^k",
|
|
587
|
+
"pass rate",
|
|
588
|
+
"score",
|
|
589
|
+
"noisy"
|
|
590
|
+
], ...rows.map((r) => [
|
|
591
|
+
r.skill,
|
|
592
|
+
r.harness,
|
|
593
|
+
String(r.scenarios),
|
|
594
|
+
`${r.pass_all}/${r.scenarios}`,
|
|
595
|
+
r.pass_rate.toFixed(2),
|
|
596
|
+
r.score.toFixed(2),
|
|
597
|
+
String(r.noisy.length)
|
|
598
|
+
])];
|
|
599
|
+
const widths = table[0].map((_, i) => Math.max(...table.map((row) => row[i].length)));
|
|
600
|
+
return table.map((row) => row.map((c, i) => c.padEnd(widths[i])).join(" ").trimEnd()).join("\n");
|
|
395
601
|
}
|
|
396
602
|
function cmdLint(argv) {
|
|
397
603
|
const { positional, flags } = parseArgs(argv);
|
|
@@ -427,4 +633,4 @@ if (isMainModule()) {
|
|
|
427
633
|
}
|
|
428
634
|
}
|
|
429
635
|
//#endregion
|
|
430
|
-
export { classifyResult, mergeScorecard, parseArgs,
|
|
636
|
+
export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, formatStats, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
|
package/dist/scenario.js
CHANGED
|
@@ -4,6 +4,7 @@ import path from "node:path";
|
|
|
4
4
|
import { createHash } from "node:crypto";
|
|
5
5
|
import os from "node:os";
|
|
6
6
|
//#region src/scenario.ts
|
|
7
|
+
const DEFAULT_CLAUDE_AGENT = "claude-opus-5";
|
|
7
8
|
function loadScenario(scenarioDir) {
|
|
8
9
|
const match = scenarioDir.match(/skills\/([^/]+)\/evals\/([^/]+)$/);
|
|
9
10
|
if (!match) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
|
|
@@ -40,7 +41,13 @@ function runNameFor(scenarioDir, harness) {
|
|
|
40
41
|
const m = path.resolve(scenarioDir).match(/skills\/([^/]+)\/evals\/([^/]+)$/);
|
|
41
42
|
if (!m) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
|
|
42
43
|
const name = `${encodeRunNamePart(m[1])}--${encodeRunNamePart(m[2])}`;
|
|
43
|
-
|
|
44
|
+
const full = harness === "claude" ? name : `${name}--${harness}`;
|
|
45
|
+
if (Buffer.byteLength(`${full}.json.attempt`) <= 255) return full;
|
|
46
|
+
return `~v3~${createHash("sha256").update(JSON.stringify([
|
|
47
|
+
m[1],
|
|
48
|
+
m[2],
|
|
49
|
+
harness
|
|
50
|
+
])).digest("hex")}`;
|
|
44
51
|
}
|
|
45
52
|
function frontmatterRange(text) {
|
|
46
53
|
const lines = text.split("\n");
|
|
@@ -140,6 +147,7 @@ function agentProvider(opts, workdir, skill, paths) {
|
|
|
140
147
|
id: "anthropic:claude-agent-sdk",
|
|
141
148
|
config: {
|
|
142
149
|
model: opts.agentModel ?? "claude-opus-5",
|
|
150
|
+
...opts.agentEffort ? { effort: opts.agentEffort } : {},
|
|
143
151
|
apiKeyRequired: false,
|
|
144
152
|
working_dir: workdir,
|
|
145
153
|
setting_sources: ["project"],
|
|
@@ -156,11 +164,17 @@ function agentProvider(opts, workdir, skill, paths) {
|
|
|
156
164
|
}
|
|
157
165
|
};
|
|
158
166
|
}
|
|
159
|
-
function
|
|
167
|
+
function trialLabel(index) {
|
|
168
|
+
return `trial-${index + 1}`;
|
|
169
|
+
}
|
|
170
|
+
function buildConfig(s, trials, opts, paths) {
|
|
160
171
|
return {
|
|
161
172
|
description: `${s.skill}/${s.scenario}`,
|
|
162
173
|
prompts: ["{{task}}"],
|
|
163
|
-
providers:
|
|
174
|
+
providers: trials.map((t, i) => ({
|
|
175
|
+
...agentProvider(opts, t.workdir, s.skill, paths),
|
|
176
|
+
label: trialLabel(i)
|
|
177
|
+
})),
|
|
164
178
|
defaultTest: { options: {
|
|
165
179
|
provider: opts.judgeModel.includes(":") ? opts.judgeEffort === void 0 ? opts.judgeModel : {
|
|
166
180
|
id: opts.judgeModel,
|
|
@@ -196,12 +210,13 @@ function buildConfig(s, workdir, manifestPath, opts, paths) {
|
|
|
196
210
|
},
|
|
197
211
|
transform: `file://${paths.transformPath}`
|
|
198
212
|
} },
|
|
199
|
-
tests:
|
|
213
|
+
tests: trials.map((t, i) => ({
|
|
200
214
|
description: s.criteria.context,
|
|
215
|
+
providers: [trialLabel(i)],
|
|
201
216
|
vars: {
|
|
202
217
|
task: s.prompt,
|
|
203
|
-
workdir,
|
|
204
|
-
manifest: manifestPath
|
|
218
|
+
workdir: t.workdir,
|
|
219
|
+
manifest: t.manifestPath
|
|
205
220
|
},
|
|
206
221
|
assert: [{
|
|
207
222
|
type: "assert-set",
|
|
@@ -215,7 +230,7 @@ function buildConfig(s, workdir, manifestPath, opts, paths) {
|
|
|
215
230
|
type: "skill-used",
|
|
216
231
|
value: s.skill
|
|
217
232
|
}]
|
|
218
|
-
}
|
|
233
|
+
}))
|
|
219
234
|
};
|
|
220
235
|
}
|
|
221
236
|
const SDK_PACKAGES = ["@anthropic-ai/claude-agent-sdk", "@openai/codex-sdk"];
|
|
@@ -244,7 +259,11 @@ function generateRun(scenarioDir, opts, paths) {
|
|
|
244
259
|
const s = loadScenario(scenarioDir);
|
|
245
260
|
const name = runNameFor(scenarioDir, opts.harness);
|
|
246
261
|
const runDir = path.join(paths.scratchDir, name);
|
|
247
|
-
|
|
262
|
+
fs.rmSync(runDir, {
|
|
263
|
+
recursive: true,
|
|
264
|
+
force: true
|
|
265
|
+
});
|
|
266
|
+
const trials = Array.from({ length: opts.trials ?? 1 }, (_, i) => materialize(s, path.join(runDir, trialLabel(i)), opts.harness));
|
|
248
267
|
const sdkDir = sdkNodeModulesDir();
|
|
249
268
|
if (sdkDir !== void 0) {
|
|
250
269
|
const link = path.join(runDir, "node_modules");
|
|
@@ -254,13 +273,15 @@ function generateRun(scenarioDir, opts, paths) {
|
|
|
254
273
|
});
|
|
255
274
|
fs.symlinkSync(sdkDir, link, "dir");
|
|
256
275
|
}
|
|
257
|
-
const config = buildConfig(s,
|
|
276
|
+
const config = buildConfig(s, trials, opts, paths);
|
|
258
277
|
const configPath = path.join(runDir, "promptfooconfig.json");
|
|
259
278
|
fs.writeFileSync(configPath, JSON.stringify(config, null, 2));
|
|
260
279
|
return {
|
|
261
280
|
name,
|
|
262
|
-
configPath
|
|
281
|
+
configPath,
|
|
282
|
+
skill: s.skill,
|
|
283
|
+
scenario: s.scenario
|
|
263
284
|
};
|
|
264
285
|
}
|
|
265
286
|
//#endregion
|
|
266
|
-
export { buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag };
|
|
287
|
+
export { DEFAULT_CLAUDE_AGENT, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
|
package/docs/adoption.md
CHANGED
|
@@ -66,9 +66,11 @@ Commit `.skillcheck/scorecards/`. Gitignore the rest:
|
|
|
66
66
|
.skillcheck/scratch/
|
|
67
67
|
```
|
|
68
68
|
|
|
69
|
-
A scorecard is only comparable against the tree it graded
|
|
70
|
-
result carries the root repo's HEAD
|
|
71
|
-
|
|
69
|
+
A scorecard is only comparable against the tree it graded and the configuration
|
|
70
|
+
it ran with, which is why every result carries the root repo's HEAD, the agent
|
|
71
|
+
and judge models and efforts, and the trial count, and why `summarize` refuses
|
|
72
|
+
to mix them without `--allow-mixed`. Use `--trials 3` or more before calling a
|
|
73
|
+
skill change better or worse; one trial cannot tell a regression from noise.
|
|
72
74
|
|
|
73
75
|
## Upgrading
|
|
74
76
|
|
package/docs/scenarios.md
CHANGED
|
@@ -54,7 +54,9 @@ item needs a non-empty `name` and `description` and a positive `max_score`.
|
|
|
54
54
|
Each item becomes one `llm-rubric` assertion weighted by `max_score`, inside an
|
|
55
55
|
assert-set with threshold 0.7. A separate `skill-used` assertion sits outside
|
|
56
56
|
that aggregate, so a run that produces good output without ever loading the
|
|
57
|
-
skill still fails. There is no test-level threshold: both must pass.
|
|
57
|
+
skill still fails. There is no test-level threshold: both must pass. The
|
|
58
|
+
reported score is the assert-set's weighted score; skill-used is reported
|
|
59
|
+
separately as a rate across trials.
|
|
58
60
|
|
|
59
61
|
Write descriptions a judge can check against the deliverable: an observable
|
|
60
62
|
property, not a feeling. Weight the items that would make a reviewer reject the
|
package/docs/usage.md
CHANGED
|
@@ -35,9 +35,10 @@ skillcheck run skills/<skill>/evals/<scenario>
|
|
|
35
35
|
skillcheck run <scenario-dir> --agent MODEL --judge MODEL --harness codex
|
|
36
36
|
skillcheck run <scenario-dir> --harness claude --max-turns 80
|
|
37
37
|
skillcheck run <scenario-dir> --harness grok
|
|
38
|
+
skillcheck run <scenario-dir> --trials 3 --agent-effort medium
|
|
38
39
|
```
|
|
39
40
|
|
|
40
|
-
Materializes the scenario into `<root>/.skillcheck/scratch/<name>/workdir`,
|
|
41
|
+
Materializes the scenario into `<root>/.skillcheck/scratch/<name>/trial-<n>/workdir`,
|
|
41
42
|
installs the skill under test into that workdir, drives the agent, and grades
|
|
42
43
|
the files it wrote. Exit 0 means pass, 1 means graded fail, and 2 means error.
|
|
43
44
|
Exit 2 covers missing usable promptfoo output or optional eval peers. The
|
|
@@ -53,6 +54,44 @@ and a Claude agent limit of 50 turns. `--max-turns` changes that limit only for
|
|
|
53
54
|
Claude; passing it with `codex` or `grok` fails before the eval starts. On
|
|
54
55
|
those harnesses, omitting `--agent` leaves the model to that CLI's own default.
|
|
55
56
|
|
|
57
|
+
### Trials
|
|
58
|
+
|
|
59
|
+
One trial is one sample of a noisy process: the same scenario and skill can
|
|
60
|
+
score 0.49 and then 0.99. `--trials <k>` (default 1) runs the agent k times,
|
|
61
|
+
each in its own workdir with its own manifest, all graded in one promptfoo eval.
|
|
62
|
+
promptfoo's `--repeat` is not used because it reuses one set of vars and one
|
|
63
|
+
`working_dir`, so concurrent trials would write into the same tree and each
|
|
64
|
+
would be graded on all of their deliverables.
|
|
65
|
+
|
|
66
|
+
A scenario's result aggregates its trials:
|
|
67
|
+
|
|
68
|
+
| Field | Meaning |
|
|
69
|
+
| ------------------------------- | ------------------------------------------------------------------- |
|
|
70
|
+
| `pass` | pass^k: every trial passed. Exit 0 needs this |
|
|
71
|
+
| `passes`, `pass_rate` | Trials that passed, as a count and a fraction |
|
|
72
|
+
| `score` | Mean weighted checklist score (the assert-set, without skill-used) |
|
|
73
|
+
| `score_min` | Lowest trial score |
|
|
74
|
+
| `score_spread` | Highest minus lowest trial score |
|
|
75
|
+
| `skill_used`, `skill_used_rate` | Trials whose `skill-used` assertion passed, as a count and fraction |
|
|
76
|
+
| `noisy` | Trials both passed and failed, or `score_spread` is at least 0.2 |
|
|
77
|
+
|
|
78
|
+
A trial that errored was never graded, so one errored trial makes the whole
|
|
79
|
+
scenario an ERROR: pass^k over fewer than k trials is not the requested number.
|
|
80
|
+
So is a result with fewer rows than trials, or a row without its checklist and
|
|
81
|
+
`skill-used` components.
|
|
82
|
+
With `--trials` above 1, `run` and `sweep` print min, spread, pass count,
|
|
83
|
+
skill-used count, and `NOISY`.
|
|
84
|
+
|
|
85
|
+
### Agent effort
|
|
86
|
+
|
|
87
|
+
`--agent-effort low|medium|high|xhigh|max` sets the Claude agent's effort.
|
|
88
|
+
promptfoo passes it to the Agent SDK, which starts Claude Code with `--effort`.
|
|
89
|
+
Omitting it leaves Claude Code's default. Only `--harness claude` takes it;
|
|
90
|
+
`codex` and `grok` fail before the eval starts. It is separate from
|
|
91
|
+
`--judge-effort`.
|
|
92
|
+
|
|
93
|
+
### Harnesses and judges
|
|
94
|
+
|
|
56
95
|
`--harness grok` runs the locally installed Grok Build CLI in the disposable
|
|
57
96
|
workdir with the skill under `.grok/skills/`. It uses native streaming events
|
|
58
97
|
to count a completed read of that skill's `SKILL.md` as `skill-used` evidence.
|
|
@@ -86,7 +125,14 @@ order, sequentially. A scenario needs both `task.md` and `criteria.json` to be
|
|
|
86
125
|
discovered. Exit 2 if anything errored, 1 if anything failed, else 0.
|
|
87
126
|
|
|
88
127
|
`EVALS_CONCURRENCY` is passed to promptfoo as `-j` (default 4). It parallelizes
|
|
89
|
-
|
|
128
|
+
the trials of one scenario, not separate scenarios. To spread scenarios, start
|
|
129
|
+
several `skillcheck run` processes; each scenario has its own result, attempt
|
|
130
|
+
marker, and scratch directory.
|
|
131
|
+
|
|
132
|
+
A scenario is skipped only when its completed result was graded with the same
|
|
133
|
+
run configuration: agent model, agent effort, judge model, judge effort, and
|
|
134
|
+
trial count. A result from another configuration is rerun and reported as
|
|
135
|
+
`RERUN`.
|
|
90
136
|
|
|
91
137
|
One known failure mode: judge calls through a gateway can drop at the transport
|
|
92
138
|
layer ([uinaf/zebroid-infra#44](https://github.com/uinaf/zebroid-infra/issues/44)).
|
|
@@ -102,7 +148,10 @@ skillcheck summarize [--allow-mixed]
|
|
|
102
148
|
|
|
103
149
|
Reduces `<root>/.skillcheck/results/*.json` into
|
|
104
150
|
`<root>/.skillcheck/scorecards/<UTC-date>.json`: one entry per scenario with
|
|
105
|
-
skill, scenario, harness, tree sha,
|
|
151
|
+
skill, scenario, harness, tree sha, the trial aggregate from [trials](#trials),
|
|
152
|
+
the run configuration, mean latency per trial, and tokens summed over trials.
|
|
153
|
+
It then prints one row per skill and harness: scenarios, pass^k count, mean pass
|
|
154
|
+
rate, mean score, and noisy count. Each noisy scenario follows on its own line.
|
|
106
155
|
|
|
107
156
|
If a scorecard for today already exists, the two are merged on
|
|
108
157
|
`(skill, scenario, harness)`: entries from this run win, entries it did not
|
|
@@ -115,14 +164,19 @@ Files that are not promptfoo results and ungraded transport errors are skipped
|
|
|
115
164
|
with a warning rather than failing the reduction. Graded assertion failures
|
|
116
165
|
remain scored results. If a skipped file matches an existing scorecard row,
|
|
117
166
|
summary generation fails and leaves the scorecard unchanged, so an errored rerun
|
|
118
|
-
cannot carry forward its old score.
|
|
167
|
+
cannot carry forward its old score. A graded result for the same identity
|
|
168
|
+
supersedes a skipped attempt only when the result file is newer. This also
|
|
169
|
+
applies with `--allow-mixed`.
|
|
119
170
|
Results from the retired Cursor harness are skipped with their original identity,
|
|
120
171
|
so they cannot become Claude scores or silently carry an old Cursor row.
|
|
121
172
|
Runs keep a `<name>.json.attempt` marker until a graded result and its provenance
|
|
122
|
-
are written. An outstanding marker makes `summarize` skip that identity even
|
|
173
|
+
are written. The marker records the original skill, scenario, and harness. An outstanding marker makes `summarize` skip that identity even
|
|
123
174
|
when the child produced no result file or left partial output. The marker does
|
|
124
175
|
not count as a result for the sweep's existence check, so no-output failures
|
|
125
176
|
remain eligible for retry.
|
|
177
|
+
Long escaped names use a short hashed filename; the original identity is kept
|
|
178
|
+
in the attempt marker and result sidecar. `sweep` retries existing results that
|
|
179
|
+
contain no grade, including results written by older versions without a marker.
|
|
126
180
|
|
|
127
181
|
## Provenance
|
|
128
182
|
|
|
@@ -131,14 +185,26 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
|
|
|
131
185
|
```json
|
|
132
186
|
{
|
|
133
187
|
"skills_tree_sha": "<root repo HEAD at run time>",
|
|
188
|
+
"skill": "<skill directory name>",
|
|
189
|
+
"scenario": "<scenario directory name>",
|
|
134
190
|
"harness": "claude",
|
|
191
|
+
"agent_model": "claude-opus-5",
|
|
192
|
+
"agent_effort": "medium",
|
|
193
|
+
"judge_model": "claude-opus-5",
|
|
194
|
+
"judge_effort": null,
|
|
195
|
+
"trials": 3,
|
|
196
|
+
"aggregate": { "pass": false, "pass_rate": 0.6667, "score": 0.81, "...": "..." },
|
|
135
197
|
"ran_at": "<ISO timestamp>",
|
|
136
198
|
"tool_version": "<skillcheck version>"
|
|
137
199
|
}
|
|
138
200
|
```
|
|
139
201
|
|
|
140
|
-
`summarize` reads those sidecars and refuses to mix skills-tree revisions
|
|
141
|
-
|
|
202
|
+
`summarize` reads those sidecars and refuses to mix skills-tree revisions or
|
|
203
|
+
run configurations (agent model and effort, judge model and effort, trials) in
|
|
204
|
+
one scorecard, including retained rows from partial reruns, unless
|
|
205
|
+
`--allow-mixed`. Configurations are compared within a harness, since harnesses
|
|
206
|
+
differ by design. Sidecars written before run configurations were recorded fall
|
|
207
|
+
back to the promptfoo config stored in the result.
|
|
142
208
|
Rejection leaves the existing scorecard unchanged. With the override, the top-level `skills_tree_sha`
|
|
143
209
|
becomes `mixed` and per-entry shas remain. A result with no sidecar reduces as
|
|
144
210
|
`unattested`.
|
|
@@ -162,3 +228,10 @@ written inside the installed package.
|
|
|
162
228
|
|
|
163
229
|
A bare `--judge` model stays on the Anthropic selection regardless of the
|
|
164
230
|
agent harness; a provider-qualified `--judge` uses that provider's env instead.
|
|
231
|
+
|
|
232
|
+
The Claude agent loads project settings only, so an `apiKeyHelper` or `env`
|
|
233
|
+
block in the operator's `~/.claude/settings.json` never reaches it; the run
|
|
234
|
+
fails with `Not logged in`. Export the gateway variables instead:
|
|
235
|
+
`ANTHROPIC_BASE_URL` and `ANTHROPIC_AUTH_TOKEN` set to the helper's output.
|
|
236
|
+
Running from inside a Claude Code session also leaks that session's
|
|
237
|
+
`CLAUDECODE` and `CLAUDE_CODE_*` variables into the agent; run from a plain shell.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@uinaf/skillcheck",
|
|
3
|
-
"version": "1.0
|
|
3
|
+
"version": "1.1.0",
|
|
4
4
|
"description": "Lint and eval harness for agent skills",
|
|
5
5
|
"homepage": "https://github.com/uinaf/skillcheck#readme",
|
|
6
6
|
"bugs": {
|
|
@@ -33,17 +33,17 @@
|
|
|
33
33
|
"prepublishOnly": "pnpm run verify:full"
|
|
34
34
|
},
|
|
35
35
|
"devDependencies": {
|
|
36
|
-
"@anthropic-ai/claude-agent-sdk": "^0.3.
|
|
37
|
-
"@openai/codex-sdk": "^0.
|
|
38
|
-
"@types/node": "^26.2
|
|
39
|
-
"promptfoo": "^0.
|
|
36
|
+
"@anthropic-ai/claude-agent-sdk": "^0.3.281",
|
|
37
|
+
"@openai/codex-sdk": "^0.156.1",
|
|
38
|
+
"@types/node": "^26.6.2",
|
|
39
|
+
"promptfoo": "^0.123.1",
|
|
40
40
|
"vite": "catalog:",
|
|
41
41
|
"vite-plus": "catalog:"
|
|
42
42
|
},
|
|
43
43
|
"peerDependencies": {
|
|
44
|
-
"@anthropic-ai/claude-agent-sdk": "^0.3.
|
|
45
|
-
"@openai/codex-sdk": "^0.
|
|
46
|
-
"promptfoo": "^0.
|
|
44
|
+
"@anthropic-ai/claude-agent-sdk": "^0.3.281",
|
|
45
|
+
"@openai/codex-sdk": "^0.156.1",
|
|
46
|
+
"promptfoo": "^0.123.1"
|
|
47
47
|
},
|
|
48
48
|
"peerDependenciesMeta": {
|
|
49
49
|
"@anthropic-ai/claude-agent-sdk": {
|
|
@@ -56,8 +56,15 @@
|
|
|
56
56
|
"optional": true
|
|
57
57
|
}
|
|
58
58
|
},
|
|
59
|
+
"devEngines": {
|
|
60
|
+
"runtime": {
|
|
61
|
+
"name": "node",
|
|
62
|
+
"version": "^24.11.0 || >=26.0.0",
|
|
63
|
+
"onFail": "error"
|
|
64
|
+
}
|
|
65
|
+
},
|
|
59
66
|
"engines": {
|
|
60
67
|
"node": ">=24"
|
|
61
68
|
},
|
|
62
|
-
"packageManager": "pnpm@12.
|
|
69
|
+
"packageManager": "pnpm@12.4.2"
|
|
63
70
|
}
|