@uinaf/skillcheck 1.0.2 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -1
- package/dist/cli.js +247 -67
- package/dist/scenario.js +27 -10
- package/docs/adoption.md +5 -3
- package/docs/scenarios.md +3 -1
- package/docs/usage.md +81 -8
- package/package.json +16 -9
package/README.md
CHANGED
|
@@ -28,7 +28,8 @@ pre-npm git tags are covered there too.
|
|
|
28
28
|
```sh
|
|
29
29
|
skillcheck lint # structural lint, no credentials
|
|
30
30
|
skillcheck run skills/wat/evals/basic # one scenario, graded end to end
|
|
31
|
-
skillcheck sweep
|
|
31
|
+
skillcheck sweep --trials 3 # every scenario, three trials each
|
|
32
|
+
skillcheck summarize # per-scenario and per-skill scorecard
|
|
32
33
|
```
|
|
33
34
|
|
|
34
35
|
`lint` needs nothing. `run` and `sweep` need model auth, which is why sweeps
|
package/dist/cli.js
CHANGED
|
@@ -12,11 +12,18 @@ const selfExt = path.extname(fileURLToPath(import.meta.url));
|
|
|
12
12
|
function toolVersion() {
|
|
13
13
|
return JSON.parse(fs.readFileSync(path.join(packageDir, "package.json"), "utf8")).version;
|
|
14
14
|
}
|
|
15
|
-
function
|
|
15
|
+
function parsePositiveInt(flag, raw) {
|
|
16
16
|
const n = Number(raw);
|
|
17
|
-
if (!Number.isInteger(n) || n <= 0) throw new Error(
|
|
17
|
+
if (!Number.isInteger(n) || n <= 0) throw new Error(`${flag} must be a positive integer, got ${JSON.stringify(raw)}`);
|
|
18
18
|
return n;
|
|
19
19
|
}
|
|
20
|
+
const AGENT_EFFORTS = [
|
|
21
|
+
"low",
|
|
22
|
+
"medium",
|
|
23
|
+
"high",
|
|
24
|
+
"xhigh",
|
|
25
|
+
"max"
|
|
26
|
+
];
|
|
20
27
|
function fail(msg) {
|
|
21
28
|
console.error(msg);
|
|
22
29
|
process.exit(1);
|
|
@@ -29,8 +36,10 @@ function parseArgs(argv) {
|
|
|
29
36
|
"--agent",
|
|
30
37
|
"--judge",
|
|
31
38
|
"--judge-effort",
|
|
39
|
+
"--agent-effort",
|
|
32
40
|
"--harness",
|
|
33
|
-
"--max-turns"
|
|
41
|
+
"--max-turns",
|
|
42
|
+
"--trials"
|
|
34
43
|
]);
|
|
35
44
|
for (let i = 0; i < argv.length; i++) {
|
|
36
45
|
const a = argv[i];
|
|
@@ -60,28 +69,67 @@ function stateDirs(root) {
|
|
|
60
69
|
}
|
|
61
70
|
function runOptions(flags) {
|
|
62
71
|
const harness = flags.get("--harness") ?? "claude";
|
|
63
|
-
if (harness !== "claude" && harness !== "codex" && harness !== "grok")
|
|
64
|
-
if (flags.has("--max-turns") && harness !== "claude")
|
|
72
|
+
if (harness !== "claude" && harness !== "codex" && harness !== "grok") throw new Error(`--harness must be claude, codex, or grok, got ${harness}`);
|
|
73
|
+
if (flags.has("--max-turns") && harness !== "claude") throw new Error("--max-turns is only supported with --harness claude");
|
|
74
|
+
const agentEffort = flags.get("--agent-effort");
|
|
75
|
+
if (agentEffort !== void 0) {
|
|
76
|
+
if (harness !== "claude") throw new Error(`--agent-effort is only supported with --harness claude; ${harness} has no effort setting wired`);
|
|
77
|
+
if (!AGENT_EFFORTS.includes(agentEffort)) throw new Error(`--agent-effort must be ${AGENT_EFFORTS.join(", ")}, got ${agentEffort}`);
|
|
78
|
+
}
|
|
65
79
|
const agent = flags.get("--agent");
|
|
66
80
|
const judgeModel = flags.get("--judge") ?? "claude-opus-5";
|
|
67
81
|
const judgeEffort = flags.get("--judge-effort");
|
|
68
82
|
if (judgeEffort !== void 0) {
|
|
69
|
-
|
|
83
|
+
const levels = judgeModel.includes(":") ? [
|
|
70
84
|
"minimal",
|
|
71
85
|
"low",
|
|
72
86
|
"medium",
|
|
73
87
|
"high"
|
|
74
|
-
]
|
|
75
|
-
if (!
|
|
88
|
+
] : AGENT_EFFORTS;
|
|
89
|
+
if (!levels.includes(judgeEffort)) throw new Error(`--judge-effort for ${judgeModel} must be ${levels.join(", ")}, got ${judgeEffort}`);
|
|
76
90
|
}
|
|
77
91
|
return {
|
|
78
92
|
harness,
|
|
79
93
|
agentModel: agent,
|
|
94
|
+
agentEffort,
|
|
80
95
|
judgeModel,
|
|
81
96
|
judgeEffort,
|
|
82
|
-
maxTurns: flags.has("--max-turns") ?
|
|
97
|
+
maxTurns: flags.has("--max-turns") ? parsePositiveInt("--max-turns", flags.get("--max-turns")) : void 0,
|
|
98
|
+
trials: flags.has("--trials") ? parsePositiveInt("--trials", flags.get("--trials")) : 1
|
|
99
|
+
};
|
|
100
|
+
}
|
|
101
|
+
function runConfigOf(opts) {
|
|
102
|
+
return {
|
|
103
|
+
agent_model: opts.agentModel ?? (opts.harness === "claude" ? "claude-opus-5" : `${opts.harness}-default`),
|
|
104
|
+
agent_effort: opts.agentEffort ?? null,
|
|
105
|
+
judge_model: opts.judgeModel,
|
|
106
|
+
judge_effort: opts.judgeEffort ?? null,
|
|
107
|
+
trials: opts.trials ?? 1
|
|
83
108
|
};
|
|
84
109
|
}
|
|
110
|
+
function configKey(c) {
|
|
111
|
+
return JSON.stringify([
|
|
112
|
+
c.agent_model,
|
|
113
|
+
c.agent_effort ?? null,
|
|
114
|
+
c.judge_model,
|
|
115
|
+
c.judge_effort ?? null,
|
|
116
|
+
c.trials ?? 1
|
|
117
|
+
]);
|
|
118
|
+
}
|
|
119
|
+
function describeConfig(c) {
|
|
120
|
+
const effort = (e) => e ? `@${e}` : "";
|
|
121
|
+
return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}`;
|
|
122
|
+
}
|
|
123
|
+
function assertUniformConfig(entries, allowMixed) {
|
|
124
|
+
if (allowMixed) return;
|
|
125
|
+
const byHarness = /* @__PURE__ */ new Map();
|
|
126
|
+
for (const e of entries) {
|
|
127
|
+
const configs = byHarness.get(e.harness) ?? /* @__PURE__ */ new Map();
|
|
128
|
+
configs.set(configKey(e), e);
|
|
129
|
+
byHarness.set(e.harness, configs);
|
|
130
|
+
}
|
|
131
|
+
for (const [harness, configs] of byHarness) if (configs.size > 1) throw new Error(`${harness} results span multiple run configurations (${[...configs.values()].map(describeConfig).join("; ")}); rerun them to match or pass --allow-mixed`);
|
|
132
|
+
}
|
|
85
133
|
function ensureEvalPackages(opts) {
|
|
86
134
|
const missing = requiredEvalPackages(opts, process.env.ANTHROPIC_API_KEY !== void 0).filter((pkg) => resolvePackageDir(pkg) === void 0);
|
|
87
135
|
if (missing.length === 0) return;
|
|
@@ -105,21 +153,76 @@ function promptfooEntry() {
|
|
|
105
153
|
if (typeof rel !== "string") throw new Error(`promptfoo at ${dir} declares no bin`);
|
|
106
154
|
return path.join(dir, rel);
|
|
107
155
|
}
|
|
108
|
-
|
|
156
|
+
const GRADED_REASONS = /* @__PURE__ */ new Set([0, 1]);
|
|
157
|
+
const ERRORED_REASON = 2;
|
|
158
|
+
function classifyResult(raw, expectedTrials) {
|
|
109
159
|
const root = raw;
|
|
110
|
-
const
|
|
160
|
+
const rows = root?.results?.results;
|
|
161
|
+
if (!Array.isArray(rows) || rows.length === 0) return { error: "promptfoo output carried no result" };
|
|
162
|
+
if (expectedTrials !== void 0 && rows.length !== expectedTrials) return { error: `promptfoo returned ${rows.length} of ${expectedTrials} trials` };
|
|
163
|
+
const stats = rows.length === 1 ? root?.results?.stats ?? void 0 : void 0;
|
|
164
|
+
const trials = [];
|
|
165
|
+
for (const [i, row] of rows.entries()) {
|
|
166
|
+
const verdict = classifyRow(row, stats);
|
|
167
|
+
if ("error" in verdict) return { error: rows.length === 1 ? verdict.error : `trial ${i + 1}: ${verdict.error}` };
|
|
168
|
+
trials.push(verdict);
|
|
169
|
+
}
|
|
170
|
+
return { trials };
|
|
171
|
+
}
|
|
172
|
+
function classifyRow(raw, stats) {
|
|
173
|
+
const res = raw;
|
|
111
174
|
if (res === void 0 || res === null) return { error: "promptfoo output carried no result" };
|
|
112
175
|
const message = typeof res.error === "string" ? res.error.trim() : "";
|
|
113
|
-
|
|
114
|
-
const
|
|
115
|
-
if (message !== "" && !
|
|
116
|
-
if (stats !== void 0 && (stats.errors ?? 0) > 0 && (stats.successes ?? 0) === 0 && (stats.failures ?? 0) === 0) return { error: "promptfoo reported an errored test with nothing graded" };
|
|
176
|
+
if (res.failureReason === ERRORED_REASON) return { error: message || "promptfoo reported an errored test" };
|
|
177
|
+
const graded = GRADED_REASONS.has(res.failureReason) || stats !== void 0 && (stats.errors ?? 0) === 0 && ((stats.failures ?? 0) > 0 || (stats.successes ?? 0) > 0);
|
|
178
|
+
if (message !== "" && !graded) return { error: message };
|
|
179
|
+
if (!graded && stats !== void 0 && (stats.errors ?? 0) > 0 && (stats.successes ?? 0) === 0 && (stats.failures ?? 0) === 0) return { error: "promptfoo reported an errored test with nothing graded" };
|
|
117
180
|
if (typeof res.score !== "number" || typeof res.success !== "boolean") return { error: "promptfoo result carried no usable score" };
|
|
181
|
+
const components = Array.isArray(res.gradingResult?.componentResults) ? res.gradingResult.componentResults : [];
|
|
182
|
+
const checklist = components.find((c) => c?.metadata?.assertionSet?.type === "assert-set");
|
|
183
|
+
const skillUsed = components.find((c) => c?.assertion?.type === "skill-used");
|
|
184
|
+
if (typeof checklist?.score !== "number" || typeof skillUsed?.pass !== "boolean") return { error: "promptfoo result carried no checklist or skill-used verdict" };
|
|
118
185
|
return {
|
|
119
|
-
score:
|
|
120
|
-
pass: res.success
|
|
186
|
+
score: checklist.score,
|
|
187
|
+
pass: res.success,
|
|
188
|
+
skillUsed: skillUsed.pass
|
|
121
189
|
};
|
|
122
190
|
}
|
|
191
|
+
const NOISY_SPREAD = .2;
|
|
192
|
+
const round4 = (n) => Math.round(n * 1e4) / 1e4;
|
|
193
|
+
function aggregateTrials(trials) {
|
|
194
|
+
if (trials.length === 0) throw new Error("cannot aggregate zero trials");
|
|
195
|
+
const scores = trials.map((t) => t.score);
|
|
196
|
+
const passes = trials.filter((t) => t.pass).length;
|
|
197
|
+
const min = Math.min(...scores);
|
|
198
|
+
const spread = Math.max(...scores) - min;
|
|
199
|
+
const skillUsed = trials.filter((t) => t.skillUsed).length;
|
|
200
|
+
return {
|
|
201
|
+
trials: trials.length,
|
|
202
|
+
pass: passes === trials.length,
|
|
203
|
+
passes,
|
|
204
|
+
pass_rate: round4(passes / trials.length),
|
|
205
|
+
score: round4(scores.reduce((a, b) => a + b, 0) / trials.length),
|
|
206
|
+
score_min: round4(min),
|
|
207
|
+
score_spread: round4(spread),
|
|
208
|
+
skill_used: skillUsed,
|
|
209
|
+
skill_used_rate: round4(skillUsed / trials.length),
|
|
210
|
+
noisy: passes > 0 && passes < trials.length || spread >= .199999999
|
|
211
|
+
};
|
|
212
|
+
}
|
|
213
|
+
function formatStats(s) {
|
|
214
|
+
const line = `score=${s.score.toFixed(4)}`;
|
|
215
|
+
if (s.trials === 1) return line;
|
|
216
|
+
return [
|
|
217
|
+
line,
|
|
218
|
+
`min=${s.score_min.toFixed(4)}`,
|
|
219
|
+
`spread=${s.score_spread.toFixed(4)}`,
|
|
220
|
+
`pass^${s.trials}=${s.pass ? "yes" : "no"}`,
|
|
221
|
+
`passes=${s.passes}/${s.trials}`,
|
|
222
|
+
`skill-used=${s.skill_used}/${s.trials}`,
|
|
223
|
+
...s.noisy ? ["NOISY"] : []
|
|
224
|
+
].join(" ");
|
|
225
|
+
}
|
|
123
226
|
function gitHead(root) {
|
|
124
227
|
return execFileSync("git", ["rev-parse", "HEAD"], {
|
|
125
228
|
cwd: root,
|
|
@@ -177,24 +280,62 @@ function runScenario(scenarioDir, opts, root) {
|
|
|
177
280
|
if (rc !== 0) return outcome;
|
|
178
281
|
let verdict;
|
|
179
282
|
try {
|
|
180
|
-
verdict = classifyResult(JSON.parse(fs.readFileSync(resultPath, "utf8")));
|
|
283
|
+
verdict = classifyResult(JSON.parse(fs.readFileSync(resultPath, "utf8")), opts.trials ?? 1);
|
|
181
284
|
} catch {
|
|
182
285
|
verdict = { error: "promptfoo produced no parseable result file" };
|
|
183
286
|
}
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
287
|
+
if ("error" in verdict) return {
|
|
288
|
+
...outcome,
|
|
289
|
+
error: verdict.error
|
|
290
|
+
};
|
|
291
|
+
outcome.stats = aggregateTrials(verdict.trials);
|
|
292
|
+
fs.writeFileSync(metaPath(resultPath), JSON.stringify({
|
|
293
|
+
skills_tree_sha: sha,
|
|
294
|
+
...identity,
|
|
295
|
+
...runConfigOf(opts),
|
|
296
|
+
aggregate: outcome.stats,
|
|
297
|
+
ran_at: (/* @__PURE__ */ new Date()).toISOString(),
|
|
298
|
+
tool_version: toolVersion()
|
|
299
|
+
}, null, 2) + "\n");
|
|
300
|
+
fs.rmSync(attemptPath(resultPath), { force: true });
|
|
196
301
|
return outcome;
|
|
197
302
|
}
|
|
303
|
+
function judgeName(judge) {
|
|
304
|
+
if (typeof judge === "string") return judge.replace(/^anthropic:messages:/, "");
|
|
305
|
+
const j = judge;
|
|
306
|
+
const name = j?.config?.model ?? j?.id;
|
|
307
|
+
if (typeof name !== "string") return "unknown";
|
|
308
|
+
return j?.config?.effort === void 0 ? name : name.replace(/^anthropic:messages:/, "");
|
|
309
|
+
}
|
|
310
|
+
function resultRunConfig(raw, meta, harness) {
|
|
311
|
+
const m = meta;
|
|
312
|
+
if (typeof m?.agent_model === "string" && typeof m.judge_model === "string") return {
|
|
313
|
+
agent_model: m.agent_model,
|
|
314
|
+
agent_effort: m.agent_effort ?? null,
|
|
315
|
+
judge_model: m.judge_model,
|
|
316
|
+
judge_effort: m.judge_effort ?? null,
|
|
317
|
+
trials: m.trials ?? 1
|
|
318
|
+
};
|
|
319
|
+
const r = raw;
|
|
320
|
+
const agent = r?.config?.providers?.[0]?.config;
|
|
321
|
+
const judge = r?.config?.defaultTest?.options?.provider;
|
|
322
|
+
const judgeConfig = judge?.config;
|
|
323
|
+
const judgeEffort = judgeConfig?.reasoning_effort ?? judgeConfig?.effort;
|
|
324
|
+
return {
|
|
325
|
+
agent_model: typeof agent?.model === "string" ? agent.model : `${harness}-default`,
|
|
326
|
+
agent_effort: typeof agent?.effort === "string" ? agent.effort : null,
|
|
327
|
+
judge_model: judgeName(judge),
|
|
328
|
+
judge_effort: typeof judgeEffort === "string" ? judgeEffort : null,
|
|
329
|
+
trials: r?.results?.results?.length ?? 1
|
|
330
|
+
};
|
|
331
|
+
}
|
|
332
|
+
function readJson(file) {
|
|
333
|
+
try {
|
|
334
|
+
return JSON.parse(fs.readFileSync(file, "utf8"));
|
|
335
|
+
} catch {
|
|
336
|
+
return;
|
|
337
|
+
}
|
|
338
|
+
}
|
|
198
339
|
function discoverScenarios(root) {
|
|
199
340
|
const roots = [path.join(root, "skills")];
|
|
200
341
|
const cliDir = path.join(root, "cli");
|
|
@@ -215,48 +356,54 @@ function discoverScenarios(root) {
|
|
|
215
356
|
}
|
|
216
357
|
return found.sort();
|
|
217
358
|
}
|
|
359
|
+
const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N]";
|
|
218
360
|
function cmdRun(argv) {
|
|
219
361
|
const { positional, flags } = parseArgs(argv);
|
|
220
|
-
if (positional.length !== 1) fail(
|
|
362
|
+
if (positional.length !== 1) fail(`usage: skillcheck run <scenario-dir> ${RUN_FLAGS}`);
|
|
221
363
|
const opts = runOptions(flags);
|
|
222
364
|
ensureEvalPackages(opts);
|
|
223
365
|
const o = runScenario(positional[0], opts, resolveRoot(flags));
|
|
224
|
-
if (o.
|
|
366
|
+
if (o.stats === void 0) {
|
|
225
367
|
console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
|
|
226
368
|
process.exit(2);
|
|
227
369
|
}
|
|
228
|
-
console.log(`${o.pass ? "PASS" : "FAIL"} ${o.name}
|
|
229
|
-
process.exit(o.pass ? 0 : 1);
|
|
370
|
+
console.log(`${o.stats.pass ? "PASS" : "FAIL"} ${o.name} ${formatStats(o.stats)} (results: ${o.resultPath})`);
|
|
371
|
+
process.exit(o.stats.pass ? 0 : 1);
|
|
230
372
|
}
|
|
231
373
|
function cmdSweep(argv) {
|
|
232
374
|
const { positional, flags } = parseArgs(argv);
|
|
233
|
-
if (positional.length > 0) fail(
|
|
375
|
+
if (positional.length > 0) fail(`usage: skillcheck sweep ${RUN_FLAGS} [--all]`);
|
|
234
376
|
const root = resolveRoot(flags);
|
|
235
377
|
const opts = runOptions(flags);
|
|
236
378
|
ensureEvalPackages(opts);
|
|
237
379
|
const all = flags.get("--all") === true;
|
|
238
380
|
const resultsDir = stateDirs(root).results;
|
|
381
|
+
const wanted = configKey(runConfigOf(opts));
|
|
239
382
|
let passed = 0, failed = 0, errored = 0, skipped = 0;
|
|
240
383
|
for (const dir of discoverScenarios(root)) {
|
|
241
384
|
const name = runNameFor(dir, opts.harness);
|
|
242
385
|
const resultPath = path.join(resultsDir, `${name}.json`);
|
|
243
|
-
if (!all && fs.existsSync(resultPath) && !fs.existsSync(attemptPath(resultPath)))
|
|
244
|
-
|
|
386
|
+
if (!all && fs.existsSync(resultPath) && !fs.existsSync(attemptPath(resultPath))) {
|
|
387
|
+
const raw = readJson(resultPath);
|
|
388
|
+
const config = resultRunConfig(raw, readJson(metaPath(resultPath)), opts.harness);
|
|
389
|
+
const graded = !("error" in classifyResult(raw, config.trials));
|
|
390
|
+
if (graded && configKey(config) === wanted) {
|
|
245
391
|
skipped++;
|
|
246
392
|
console.log(`SKIP ${name} (results exist; use --all to rerun)`);
|
|
247
393
|
continue;
|
|
248
394
|
}
|
|
249
|
-
|
|
395
|
+
if (graded) console.log(`RERUN ${name} (results used ${describeConfig(config)})`);
|
|
396
|
+
}
|
|
250
397
|
const o = runScenario(dir, opts, root);
|
|
251
|
-
if (o.
|
|
398
|
+
if (o.stats === void 0) {
|
|
252
399
|
errored++;
|
|
253
400
|
console.log(`ERROR ${o.name} ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
|
|
254
|
-
} else if (o.pass) {
|
|
401
|
+
} else if (o.stats.pass) {
|
|
255
402
|
passed++;
|
|
256
|
-
console.log(`PASS ${o.name}
|
|
403
|
+
console.log(`PASS ${o.name} ${formatStats(o.stats)}`);
|
|
257
404
|
} else {
|
|
258
405
|
failed++;
|
|
259
|
-
console.log(`FAIL ${o.name}
|
|
406
|
+
console.log(`FAIL ${o.name} ${formatStats(o.stats)}`);
|
|
260
407
|
}
|
|
261
408
|
}
|
|
262
409
|
console.log(`\nsweep: ${passed} passed, ${failed} failed, ${errored} errored, ${skipped} skipped`);
|
|
@@ -310,45 +457,36 @@ function reduceResults(dir, allowMixed) {
|
|
|
310
457
|
skipped.push(f);
|
|
311
458
|
continue;
|
|
312
459
|
}
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
}
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
const verdict = classifyResult(raw);
|
|
321
|
-
if (verdict.score === void 0 || verdict.pass === void 0) {
|
|
460
|
+
const raw = readJson(path.join(dir, f));
|
|
461
|
+
const base = f.replace(/\.json$/, "");
|
|
462
|
+
const meta = readJson(path.join(dir, `${base}.meta.json`));
|
|
463
|
+
const { skill, scenario, harness } = resultIdentity(f, dir);
|
|
464
|
+
const config = resultRunConfig(raw, meta, harness);
|
|
465
|
+
const verdict = classifyResult(raw, config.trials);
|
|
466
|
+
if ("error" in verdict) {
|
|
322
467
|
console.error(`skipping ${f}: ${verdict.error}`);
|
|
323
468
|
skipped.push(f);
|
|
324
469
|
continue;
|
|
325
470
|
}
|
|
326
|
-
const
|
|
327
|
-
const judge = raw.config?.defaultTest?.options?.provider;
|
|
328
|
-
const base = f.replace(/\.json$/, "");
|
|
329
|
-
const { skill, scenario, harness } = resultIdentity(f, dir);
|
|
471
|
+
const rows = raw.results.results;
|
|
330
472
|
const key = entryKey({
|
|
331
473
|
skill,
|
|
332
474
|
scenario,
|
|
333
475
|
harness
|
|
334
476
|
});
|
|
335
477
|
gradedAt.set(key, Math.max(gradedAt.get(key) ?? 0, fs.statSync(path.join(dir, f)).mtimeMs));
|
|
336
|
-
|
|
337
|
-
try {
|
|
338
|
-
sha = JSON.parse(fs.readFileSync(path.join(dir, `${base}.meta.json`), "utf8")).skills_tree_sha ?? "unattested";
|
|
339
|
-
} catch {}
|
|
478
|
+
const sha = typeof meta?.skills_tree_sha === "string" ? meta.skills_tree_sha : "unattested";
|
|
340
479
|
shas.add(sha);
|
|
480
|
+
const stats = aggregateTrials(verdict.trials);
|
|
341
481
|
entries.push({
|
|
342
482
|
skill,
|
|
343
483
|
scenario,
|
|
344
484
|
harness,
|
|
345
485
|
skills_tree_sha: sha,
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
latency_ms: res.latencyMs,
|
|
351
|
-
tokens: (res.tokenUsage?.total ?? 0) + (res.tokenUsage?.assertions?.total ?? 0)
|
|
486
|
+
...stats,
|
|
487
|
+
...config,
|
|
488
|
+
latency_ms: Math.round(rows.reduce((a, r) => a + (r.latencyMs ?? 0), 0) / rows.length),
|
|
489
|
+
tokens: rows.reduce((a, r) => a + (r.tokenUsage?.total ?? 0) + (r.tokenUsage?.assertions?.total ?? 0), 0)
|
|
352
490
|
});
|
|
353
491
|
}
|
|
354
492
|
if (shas.size > 1 && !allowMixed) throw new Error(`results span multiple skills-tree revisions (${[...shas].join(", ")}); rerun stale ones or pass --allow-mixed`);
|
|
@@ -411,7 +549,9 @@ function cmdSummarize(argv) {
|
|
|
411
549
|
if (existing.some((entry) => skippedKeys.has(entryKey(entry)))) throw new Error("skipped rerun matches an existing score; refusing to carry it or overwrite the scorecard");
|
|
412
550
|
const merged = mergeScorecard(existing, entries);
|
|
413
551
|
const treeSha = treeShaOf(merged.entries);
|
|
414
|
-
|
|
552
|
+
const allowMixed = flags.get("--allow-mixed") === true;
|
|
553
|
+
if (treeSha === "mixed" && !allowMixed) throw new Error("scorecard spans multiple skills-tree revisions; rerun stale ones or pass --allow-mixed");
|
|
554
|
+
assertUniformConfig(merged.entries, allowMixed);
|
|
415
555
|
const scorecard = {
|
|
416
556
|
ran_at: (/* @__PURE__ */ new Date()).toISOString(),
|
|
417
557
|
skills_tree_sha: treeSha,
|
|
@@ -420,6 +560,46 @@ function cmdSummarize(argv) {
|
|
|
420
560
|
fs.writeFileSync(out, JSON.stringify(scorecard, null, 2) + "\n");
|
|
421
561
|
console.log(`${out}: ${merged.entries.length} scenario(s), ${merged.entries.filter((e) => e.pass).length} passing, ${skipped.length} skipped file(s)`);
|
|
422
562
|
if (existing.length > 0) console.log(`merged into today's scorecard: ${entries.length} from this run, ${merged.carried} carried over`);
|
|
563
|
+
console.log(`\n${formatSkillTable(summarizeSkills(merged.entries))}`);
|
|
564
|
+
for (const e of merged.entries.filter((x) => x.noisy)) console.log(`NOISY ${e.skill}/${e.scenario} (${e.harness}) ${formatStats(e)}`);
|
|
565
|
+
}
|
|
566
|
+
function summarizeSkills(entries) {
|
|
567
|
+
const groups = /* @__PURE__ */ new Map();
|
|
568
|
+
for (const e of entries) {
|
|
569
|
+
const key = `${e.skill}\0${e.harness}`;
|
|
570
|
+
groups.set(key, [...groups.get(key) ?? [], e]);
|
|
571
|
+
}
|
|
572
|
+
const mean = (xs) => round4(xs.reduce((a, b) => a + b, 0) / xs.length);
|
|
573
|
+
return [...groups.values()].map((rows) => ({
|
|
574
|
+
skill: rows[0].skill,
|
|
575
|
+
harness: rows[0].harness,
|
|
576
|
+
scenarios: rows.length,
|
|
577
|
+
pass_all: rows.filter((r) => r.pass).length,
|
|
578
|
+
pass_rate: mean(rows.map((r) => r.pass_rate ?? (r.pass ? 1 : 0))),
|
|
579
|
+
score: mean(rows.map((r) => r.score)),
|
|
580
|
+
noisy: rows.filter((r) => r.noisy).map((r) => r.scenario)
|
|
581
|
+
}));
|
|
582
|
+
}
|
|
583
|
+
function formatSkillTable(rows) {
|
|
584
|
+
const table = [[
|
|
585
|
+
"skill",
|
|
586
|
+
"harness",
|
|
587
|
+
"scenarios",
|
|
588
|
+
"pass^k",
|
|
589
|
+
"pass rate",
|
|
590
|
+
"score",
|
|
591
|
+
"noisy"
|
|
592
|
+
], ...rows.map((r) => [
|
|
593
|
+
r.skill,
|
|
594
|
+
r.harness,
|
|
595
|
+
String(r.scenarios),
|
|
596
|
+
`${r.pass_all}/${r.scenarios}`,
|
|
597
|
+
r.pass_rate.toFixed(2),
|
|
598
|
+
r.score.toFixed(2),
|
|
599
|
+
String(r.noisy.length)
|
|
600
|
+
])];
|
|
601
|
+
const widths = table[0].map((_, i) => Math.max(...table.map((row) => row[i].length)));
|
|
602
|
+
return table.map((row) => row.map((c, i) => c.padEnd(widths[i])).join(" ").trimEnd()).join("\n");
|
|
423
603
|
}
|
|
424
604
|
function cmdLint(argv) {
|
|
425
605
|
const { positional, flags } = parseArgs(argv);
|
|
@@ -455,4 +635,4 @@ if (isMainModule()) {
|
|
|
455
635
|
}
|
|
456
636
|
}
|
|
457
637
|
//#endregion
|
|
458
|
-
export { classifyResult, mergeScorecard, parseArgs,
|
|
638
|
+
export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, formatStats, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
|
package/dist/scenario.js
CHANGED
|
@@ -4,6 +4,7 @@ import path from "node:path";
|
|
|
4
4
|
import { createHash } from "node:crypto";
|
|
5
5
|
import os from "node:os";
|
|
6
6
|
//#region src/scenario.ts
|
|
7
|
+
const DEFAULT_CLAUDE_AGENT = "claude-opus-5";
|
|
7
8
|
function loadScenario(scenarioDir) {
|
|
8
9
|
const match = scenarioDir.match(/skills\/([^/]+)\/evals\/([^/]+)$/);
|
|
9
10
|
if (!match) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
|
|
@@ -146,6 +147,7 @@ function agentProvider(opts, workdir, skill, paths) {
|
|
|
146
147
|
id: "anthropic:claude-agent-sdk",
|
|
147
148
|
config: {
|
|
148
149
|
model: opts.agentModel ?? "claude-opus-5",
|
|
150
|
+
...opts.agentEffort ? { effort: opts.agentEffort } : {},
|
|
149
151
|
apiKeyRequired: false,
|
|
150
152
|
working_dir: workdir,
|
|
151
153
|
setting_sources: ["project"],
|
|
@@ -162,19 +164,29 @@ function agentProvider(opts, workdir, skill, paths) {
|
|
|
162
164
|
}
|
|
163
165
|
};
|
|
164
166
|
}
|
|
165
|
-
function
|
|
167
|
+
function trialLabel(index) {
|
|
168
|
+
return `trial-${index + 1}`;
|
|
169
|
+
}
|
|
170
|
+
function buildConfig(s, trials, opts, paths) {
|
|
166
171
|
return {
|
|
167
172
|
description: `${s.skill}/${s.scenario}`,
|
|
168
173
|
prompts: ["{{task}}"],
|
|
169
|
-
providers:
|
|
174
|
+
providers: trials.map((t, i) => ({
|
|
175
|
+
...agentProvider(opts, t.workdir, s.skill, paths),
|
|
176
|
+
label: trialLabel(i)
|
|
177
|
+
})),
|
|
170
178
|
defaultTest: { options: {
|
|
171
179
|
provider: opts.judgeModel.includes(":") ? opts.judgeEffort === void 0 ? opts.judgeModel : {
|
|
172
180
|
id: opts.judgeModel,
|
|
173
181
|
config: { reasoning_effort: opts.judgeEffort }
|
|
174
|
-
} : process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
|
|
182
|
+
} : process.env.ANTHROPIC_API_KEY ? opts.judgeEffort === void 0 ? `anthropic:messages:${opts.judgeModel}` : {
|
|
183
|
+
id: `anthropic:messages:${opts.judgeModel}`,
|
|
184
|
+
config: { effort: opts.judgeEffort }
|
|
185
|
+
} : {
|
|
175
186
|
id: "anthropic:claude-agent-sdk",
|
|
176
187
|
config: {
|
|
177
188
|
model: opts.judgeModel,
|
|
189
|
+
...opts.judgeEffort ? { effort: opts.judgeEffort } : {},
|
|
178
190
|
apiKeyRequired: false,
|
|
179
191
|
max_turns: 3,
|
|
180
192
|
output_format: {
|
|
@@ -202,12 +214,13 @@ function buildConfig(s, workdir, manifestPath, opts, paths) {
|
|
|
202
214
|
},
|
|
203
215
|
transform: `file://${paths.transformPath}`
|
|
204
216
|
} },
|
|
205
|
-
tests:
|
|
217
|
+
tests: trials.map((t, i) => ({
|
|
206
218
|
description: s.criteria.context,
|
|
219
|
+
providers: [trialLabel(i)],
|
|
207
220
|
vars: {
|
|
208
221
|
task: s.prompt,
|
|
209
|
-
workdir,
|
|
210
|
-
manifest: manifestPath
|
|
222
|
+
workdir: t.workdir,
|
|
223
|
+
manifest: t.manifestPath
|
|
211
224
|
},
|
|
212
225
|
assert: [{
|
|
213
226
|
type: "assert-set",
|
|
@@ -221,7 +234,7 @@ function buildConfig(s, workdir, manifestPath, opts, paths) {
|
|
|
221
234
|
type: "skill-used",
|
|
222
235
|
value: s.skill
|
|
223
236
|
}]
|
|
224
|
-
}
|
|
237
|
+
}))
|
|
225
238
|
};
|
|
226
239
|
}
|
|
227
240
|
const SDK_PACKAGES = ["@anthropic-ai/claude-agent-sdk", "@openai/codex-sdk"];
|
|
@@ -250,7 +263,11 @@ function generateRun(scenarioDir, opts, paths) {
|
|
|
250
263
|
const s = loadScenario(scenarioDir);
|
|
251
264
|
const name = runNameFor(scenarioDir, opts.harness);
|
|
252
265
|
const runDir = path.join(paths.scratchDir, name);
|
|
253
|
-
|
|
266
|
+
fs.rmSync(runDir, {
|
|
267
|
+
recursive: true,
|
|
268
|
+
force: true
|
|
269
|
+
});
|
|
270
|
+
const trials = Array.from({ length: opts.trials ?? 1 }, (_, i) => materialize(s, path.join(runDir, trialLabel(i)), opts.harness));
|
|
254
271
|
const sdkDir = sdkNodeModulesDir();
|
|
255
272
|
if (sdkDir !== void 0) {
|
|
256
273
|
const link = path.join(runDir, "node_modules");
|
|
@@ -260,7 +277,7 @@ function generateRun(scenarioDir, opts, paths) {
|
|
|
260
277
|
});
|
|
261
278
|
fs.symlinkSync(sdkDir, link, "dir");
|
|
262
279
|
}
|
|
263
|
-
const config = buildConfig(s,
|
|
280
|
+
const config = buildConfig(s, trials, opts, paths);
|
|
264
281
|
const configPath = path.join(runDir, "promptfooconfig.json");
|
|
265
282
|
fs.writeFileSync(configPath, JSON.stringify(config, null, 2));
|
|
266
283
|
return {
|
|
@@ -271,4 +288,4 @@ function generateRun(scenarioDir, opts, paths) {
|
|
|
271
288
|
};
|
|
272
289
|
}
|
|
273
290
|
//#endregion
|
|
274
|
-
export { buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag };
|
|
291
|
+
export { DEFAULT_CLAUDE_AGENT, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
|
package/docs/adoption.md
CHANGED
|
@@ -66,9 +66,11 @@ Commit `.skillcheck/scorecards/`. Gitignore the rest:
|
|
|
66
66
|
.skillcheck/scratch/
|
|
67
67
|
```
|
|
68
68
|
|
|
69
|
-
A scorecard is only comparable against the tree it graded
|
|
70
|
-
result carries the root repo's HEAD
|
|
71
|
-
|
|
69
|
+
A scorecard is only comparable against the tree it graded and the configuration
|
|
70
|
+
it ran with, which is why every result carries the root repo's HEAD, the agent
|
|
71
|
+
and judge models and efforts, and the trial count, and why `summarize` refuses
|
|
72
|
+
to mix them without `--allow-mixed`. Use `--trials 3` or more before calling a
|
|
73
|
+
skill change better or worse; one trial cannot tell a regression from noise.
|
|
72
74
|
|
|
73
75
|
## Upgrading
|
|
74
76
|
|
package/docs/scenarios.md
CHANGED
|
@@ -54,7 +54,9 @@ item needs a non-empty `name` and `description` and a positive `max_score`.
|
|
|
54
54
|
Each item becomes one `llm-rubric` assertion weighted by `max_score`, inside an
|
|
55
55
|
assert-set with threshold 0.7. A separate `skill-used` assertion sits outside
|
|
56
56
|
that aggregate, so a run that produces good output without ever loading the
|
|
57
|
-
skill still fails. There is no test-level threshold: both must pass.
|
|
57
|
+
skill still fails. There is no test-level threshold: both must pass. The
|
|
58
|
+
reported score is the assert-set's weighted score; skill-used is reported
|
|
59
|
+
separately as a rate across trials.
|
|
58
60
|
|
|
59
61
|
Write descriptions a judge can check against the deliverable: an observable
|
|
60
62
|
property, not a feeling. Weight the items that would make a reviewer reject the
|
package/docs/usage.md
CHANGED
|
@@ -35,9 +35,10 @@ skillcheck run skills/<skill>/evals/<scenario>
|
|
|
35
35
|
skillcheck run <scenario-dir> --agent MODEL --judge MODEL --harness codex
|
|
36
36
|
skillcheck run <scenario-dir> --harness claude --max-turns 80
|
|
37
37
|
skillcheck run <scenario-dir> --harness grok
|
|
38
|
+
skillcheck run <scenario-dir> --trials 3 --agent-effort medium
|
|
38
39
|
```
|
|
39
40
|
|
|
40
|
-
Materializes the scenario into `<root>/.skillcheck/scratch/<name>/workdir`,
|
|
41
|
+
Materializes the scenario into `<root>/.skillcheck/scratch/<name>/trial-<n>/workdir`,
|
|
41
42
|
installs the skill under test into that workdir, drives the agent, and grades
|
|
42
43
|
the files it wrote. Exit 0 means pass, 1 means graded fail, and 2 means error.
|
|
43
44
|
Exit 2 covers missing usable promptfoo output or optional eval peers. The
|
|
@@ -53,6 +54,44 @@ and a Claude agent limit of 50 turns. `--max-turns` changes that limit only for
|
|
|
53
54
|
Claude; passing it with `codex` or `grok` fails before the eval starts. On
|
|
54
55
|
those harnesses, omitting `--agent` leaves the model to that CLI's own default.
|
|
55
56
|
|
|
57
|
+
### Trials
|
|
58
|
+
|
|
59
|
+
One trial is one sample of a noisy process: the same scenario and skill can
|
|
60
|
+
score 0.49 and then 0.99. `--trials <k>` (default 1) runs the agent k times,
|
|
61
|
+
each in its own workdir with its own manifest, all graded in one promptfoo eval.
|
|
62
|
+
promptfoo's `--repeat` is not used because it reuses one set of vars and one
|
|
63
|
+
`working_dir`, so concurrent trials would write into the same tree and each
|
|
64
|
+
would be graded on all of their deliverables.
|
|
65
|
+
|
|
66
|
+
A scenario's result aggregates its trials:
|
|
67
|
+
|
|
68
|
+
| Field | Meaning |
|
|
69
|
+
| ------------------------------- | ------------------------------------------------------------------- |
|
|
70
|
+
| `pass` | pass^k: every trial passed. Exit 0 needs this |
|
|
71
|
+
| `passes`, `pass_rate` | Trials that passed, as a count and a fraction |
|
|
72
|
+
| `score` | Mean weighted checklist score (the assert-set, without skill-used) |
|
|
73
|
+
| `score_min` | Lowest trial score |
|
|
74
|
+
| `score_spread` | Highest minus lowest trial score |
|
|
75
|
+
| `skill_used`, `skill_used_rate` | Trials whose `skill-used` assertion passed, as a count and fraction |
|
|
76
|
+
| `noisy` | Trials both passed and failed, or `score_spread` is at least 0.2 |
|
|
77
|
+
|
|
78
|
+
A trial that errored was never graded, so one errored trial makes the whole
|
|
79
|
+
scenario an ERROR: pass^k over fewer than k trials is not the requested number.
|
|
80
|
+
So is a result with fewer rows than trials, or a row without its checklist and
|
|
81
|
+
`skill-used` components.
|
|
82
|
+
With `--trials` above 1, `run` and `sweep` print min, spread, pass count,
|
|
83
|
+
skill-used count, and `NOISY`.
|
|
84
|
+
|
|
85
|
+
### Agent effort
|
|
86
|
+
|
|
87
|
+
`--agent-effort low|medium|high|xhigh|max` sets the Claude agent's effort.
|
|
88
|
+
promptfoo passes it to the Agent SDK, which starts Claude Code with `--effort`.
|
|
89
|
+
Omitting it leaves Claude Code's default. Only `--harness claude` takes it;
|
|
90
|
+
`codex` and `grok` fail before the eval starts. It is separate from
|
|
91
|
+
`--judge-effort`.
|
|
92
|
+
|
|
93
|
+
### Harnesses and judges
|
|
94
|
+
|
|
56
95
|
`--harness grok` runs the locally installed Grok Build CLI in the disposable
|
|
57
96
|
workdir with the skill under `.grok/skills/`. It uses native streaming events
|
|
58
97
|
to count a completed read of that skill's `SKILL.md` as `skill-used` evidence.
|
|
@@ -70,9 +109,16 @@ skillcheck run <scenario-dir> --judge openai:chat:gpt-5.6-sol --judge-effort hig
|
|
|
70
109
|
|
|
71
110
|
A provider-qualified judge authenticates through that provider's own env
|
|
72
111
|
(`OPENAI_API_KEY`, plus `OPENAI_BASE_URL` for a gateway) and is recorded
|
|
73
|
-
verbatim in the scorecard's `judge_model` column. `--judge-effort`
|
|
74
|
-
(minimal|low|medium|high) sets `reasoning_effort
|
|
75
|
-
|
|
112
|
+
verbatim in the scorecard's `judge_model` column. For it, `--judge-effort`
|
|
113
|
+
(minimal|low|medium|high) sets `reasoning_effort`. For a bare Claude judge,
|
|
114
|
+
`--judge-effort` takes Claude's levels (low|medium|high|xhigh|max) and is passed
|
|
115
|
+
as `effort` on either Anthropic path; the SDK judge starts Claude Code with
|
|
116
|
+
`--effort`:
|
|
117
|
+
|
|
118
|
+
```sh
|
|
119
|
+
skillcheck run <scenario-dir> --agent claude-opus-5-5 --agent-effort medium \
|
|
120
|
+
--judge claude-opus-5-5 --judge-effort high --trials 3
|
|
121
|
+
```
|
|
76
122
|
|
|
77
123
|
## Sweep
|
|
78
124
|
|
|
@@ -86,7 +132,14 @@ order, sequentially. A scenario needs both `task.md` and `criteria.json` to be
|
|
|
86
132
|
discovered. Exit 2 if anything errored, 1 if anything failed, else 0.
|
|
87
133
|
|
|
88
134
|
`EVALS_CONCURRENCY` is passed to promptfoo as `-j` (default 4). It parallelizes
|
|
89
|
-
|
|
135
|
+
the trials of one scenario, not separate scenarios. To spread scenarios, start
|
|
136
|
+
several `skillcheck run` processes; each scenario has its own result, attempt
|
|
137
|
+
marker, and scratch directory.
|
|
138
|
+
|
|
139
|
+
A scenario is skipped only when its completed result was graded with the same
|
|
140
|
+
run configuration: agent model, agent effort, judge model, judge effort, and
|
|
141
|
+
trial count. A result from another configuration is rerun and reported as
|
|
142
|
+
`RERUN`.
|
|
90
143
|
|
|
91
144
|
One known failure mode: judge calls through a gateway can drop at the transport
|
|
92
145
|
layer ([uinaf/zebroid-infra#44](https://github.com/uinaf/zebroid-infra/issues/44)).
|
|
@@ -102,7 +155,10 @@ skillcheck summarize [--allow-mixed]
|
|
|
102
155
|
|
|
103
156
|
Reduces `<root>/.skillcheck/results/*.json` into
|
|
104
157
|
`<root>/.skillcheck/scorecards/<UTC-date>.json`: one entry per scenario with
|
|
105
|
-
skill, scenario, harness, tree sha,
|
|
158
|
+
skill, scenario, harness, tree sha, the trial aggregate from [trials](#trials),
|
|
159
|
+
the run configuration, mean latency per trial, and tokens summed over trials.
|
|
160
|
+
It then prints one row per skill and harness: scenarios, pass^k count, mean pass
|
|
161
|
+
rate, mean score, and noisy count. Each noisy scenario follows on its own line.
|
|
106
162
|
|
|
107
163
|
If a scorecard for today already exists, the two are merged on
|
|
108
164
|
`(skill, scenario, harness)`: entries from this run win, entries it did not
|
|
@@ -139,13 +195,23 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
|
|
|
139
195
|
"skill": "<skill directory name>",
|
|
140
196
|
"scenario": "<scenario directory name>",
|
|
141
197
|
"harness": "claude",
|
|
198
|
+
"agent_model": "claude-opus-5",
|
|
199
|
+
"agent_effort": "medium",
|
|
200
|
+
"judge_model": "claude-opus-5",
|
|
201
|
+
"judge_effort": null,
|
|
202
|
+
"trials": 3,
|
|
203
|
+
"aggregate": { "pass": false, "pass_rate": 0.6667, "score": 0.81, "...": "..." },
|
|
142
204
|
"ran_at": "<ISO timestamp>",
|
|
143
205
|
"tool_version": "<skillcheck version>"
|
|
144
206
|
}
|
|
145
207
|
```
|
|
146
208
|
|
|
147
|
-
`summarize` reads those sidecars and refuses to mix skills-tree revisions
|
|
148
|
-
|
|
209
|
+
`summarize` reads those sidecars and refuses to mix skills-tree revisions or
|
|
210
|
+
run configurations (agent model and effort, judge model and effort, trials) in
|
|
211
|
+
one scorecard, including retained rows from partial reruns, unless
|
|
212
|
+
`--allow-mixed`. Configurations are compared within a harness, since harnesses
|
|
213
|
+
differ by design. Sidecars written before run configurations were recorded fall
|
|
214
|
+
back to the promptfoo config stored in the result.
|
|
149
215
|
Rejection leaves the existing scorecard unchanged. With the override, the top-level `skills_tree_sha`
|
|
150
216
|
becomes `mixed` and per-entry shas remain. A result with no sidecar reduces as
|
|
151
217
|
`unattested`.
|
|
@@ -169,3 +235,10 @@ written inside the installed package.
|
|
|
169
235
|
|
|
170
236
|
A bare `--judge` model stays on the Anthropic selection regardless of the
|
|
171
237
|
agent harness; a provider-qualified `--judge` uses that provider's env instead.
|
|
238
|
+
|
|
239
|
+
The Claude agent loads project settings only, so an `apiKeyHelper` or `env`
|
|
240
|
+
block in the operator's `~/.claude/settings.json` never reaches it; the run
|
|
241
|
+
fails with `Not logged in`. Export the gateway variables instead:
|
|
242
|
+
`ANTHROPIC_BASE_URL` and `ANTHROPIC_AUTH_TOKEN` set to the helper's output.
|
|
243
|
+
Running from inside a Claude Code session also leaks that session's
|
|
244
|
+
`CLAUDECODE` and `CLAUDE_CODE_*` variables into the agent; run from a plain shell.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@uinaf/skillcheck",
|
|
3
|
-
"version": "1.0
|
|
3
|
+
"version": "1.2.0",
|
|
4
4
|
"description": "Lint and eval harness for agent skills",
|
|
5
5
|
"homepage": "https://github.com/uinaf/skillcheck#readme",
|
|
6
6
|
"bugs": {
|
|
@@ -33,17 +33,17 @@
|
|
|
33
33
|
"prepublishOnly": "pnpm run verify:full"
|
|
34
34
|
},
|
|
35
35
|
"devDependencies": {
|
|
36
|
-
"@anthropic-ai/claude-agent-sdk": "^0.3.
|
|
37
|
-
"@openai/codex-sdk": "^0.
|
|
38
|
-
"@types/node": "^26.2
|
|
39
|
-
"promptfoo": "^0.
|
|
36
|
+
"@anthropic-ai/claude-agent-sdk": "^0.3.281",
|
|
37
|
+
"@openai/codex-sdk": "^0.156.1",
|
|
38
|
+
"@types/node": "^26.6.2",
|
|
39
|
+
"promptfoo": "^0.123.1",
|
|
40
40
|
"vite": "catalog:",
|
|
41
41
|
"vite-plus": "catalog:"
|
|
42
42
|
},
|
|
43
43
|
"peerDependencies": {
|
|
44
|
-
"@anthropic-ai/claude-agent-sdk": "^0.3.
|
|
45
|
-
"@openai/codex-sdk": "^0.
|
|
46
|
-
"promptfoo": "^0.
|
|
44
|
+
"@anthropic-ai/claude-agent-sdk": "^0.3.281",
|
|
45
|
+
"@openai/codex-sdk": "^0.156.1",
|
|
46
|
+
"promptfoo": "^0.123.1"
|
|
47
47
|
},
|
|
48
48
|
"peerDependenciesMeta": {
|
|
49
49
|
"@anthropic-ai/claude-agent-sdk": {
|
|
@@ -56,8 +56,15 @@
|
|
|
56
56
|
"optional": true
|
|
57
57
|
}
|
|
58
58
|
},
|
|
59
|
+
"devEngines": {
|
|
60
|
+
"runtime": {
|
|
61
|
+
"name": "node",
|
|
62
|
+
"version": "^24.11.0 || >=26.0.0",
|
|
63
|
+
"onFail": "error"
|
|
64
|
+
}
|
|
65
|
+
},
|
|
59
66
|
"engines": {
|
|
60
67
|
"node": ">=24"
|
|
61
68
|
},
|
|
62
|
-
"packageManager": "pnpm@12.
|
|
69
|
+
"packageManager": "pnpm@12.4.2"
|
|
63
70
|
}
|