@uinaf/skillcheck 0.4.0 → 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -22
- package/dist/cli.js +78 -23
- package/dist/scenario.js +19 -2
- package/docs/adoption.md +29 -19
- package/docs/authoring.md +34 -34
- package/docs/releasing.md +24 -24
- package/docs/scenarios.md +31 -31
- package/docs/usage.md +65 -54
- package/package.json +23 -8
package/README.md
CHANGED
|
@@ -2,25 +2,28 @@
|
|
|
2
2
|
|
|
3
3
|
# uinaf/skillcheck
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
Lint and eval harness for agent skills. One CLI with two halves: a keyless
|
|
6
6
|
structural lint any repo can run in CI, and a promptfoo-driven eval loop that
|
|
7
7
|
grades what a skill actually makes an agent do.
|
|
8
8
|
|
|
9
|
-
|
|
10
|
-
uinaf-specific.
|
|
9
|
+
Built for [uinaf](https://uinaf.dev) skill repos. Nothing in it is
|
|
10
|
+
uinaf-specific. It ships no opinion about what a skill should say, only about
|
|
11
11
|
where skills sit and how a scenario is scored.
|
|
12
12
|
|
|
13
|
-
##
|
|
13
|
+
## Install
|
|
14
14
|
|
|
15
15
|
```sh
|
|
16
16
|
pnpm add -D @uinaf/skillcheck
|
|
17
17
|
```
|
|
18
18
|
|
|
19
|
-
|
|
20
|
-
so a runner using `--ignore-scripts` is fine
|
|
21
|
-
|
|
19
|
+
Node 24 or newer. The package ships compiled ESM, runs no install scripts, and
|
|
20
|
+
has no regular dependencies, so a runner using `--ignore-scripts` is fine and
|
|
21
|
+
the lint-only install stays at a handful of packages. The promptfoo eval engine
|
|
22
|
+
and provider SDKs are optional peers, installed only on the operator machine
|
|
23
|
+
that runs evals ([adoption](docs/adoption.md)). Consumers still pinned to the
|
|
24
|
+
pre-npm git tags are covered there too.
|
|
22
25
|
|
|
23
|
-
##
|
|
26
|
+
## Use
|
|
24
27
|
|
|
25
28
|
```sh
|
|
26
29
|
skillcheck lint # structural lint, no credentials
|
|
@@ -29,12 +32,12 @@ skillcheck sweep && skillcheck summarize # every scenario, then a scorecard
|
|
|
29
32
|
```
|
|
30
33
|
|
|
31
34
|
`lint` needs nothing. `run` and `sweep` need model auth, which is why sweeps
|
|
32
|
-
stay operator-run and consumer repos never hold credentials.
|
|
35
|
+
stay operator-run and consumer repos never hold credentials. Every command,
|
|
33
36
|
flag, and auth variable is in [usage](docs/usage.md).
|
|
34
37
|
|
|
35
|
-
##
|
|
38
|
+
## Layout contract
|
|
36
39
|
|
|
37
|
-
|
|
40
|
+
Frozen, not configurable. Every command reads one root: `--root <dir>`, or the
|
|
38
41
|
current directory.
|
|
39
42
|
|
|
40
43
|
```text
|
|
@@ -47,18 +50,18 @@ current directory.
|
|
|
47
50
|
`cli/*/skills/<skill>/` is scanned too, for repos that keep a skill next to the
|
|
48
51
|
CLI it documents.
|
|
49
52
|
|
|
50
|
-
##
|
|
53
|
+
## Docs
|
|
51
54
|
|
|
52
|
-
|
|
|
55
|
+
| Doc | When |
|
|
53
56
|
| --------------------------------------------------------------- | ------------------------------------- |
|
|
54
|
-
| [
|
|
55
|
-
| [
|
|
56
|
-
| [
|
|
57
|
-
| [
|
|
58
|
-
| [
|
|
59
|
-
| [
|
|
60
|
-
| [
|
|
61
|
-
|
|
62
|
-
##
|
|
57
|
+
| [Usage](docs/usage.md) | Every subcommand, flag, and auth path |
|
|
58
|
+
| [Scenarios](docs/scenarios.md) | Writing an eval scenario |
|
|
59
|
+
| [Authoring](docs/authoring.md) | Writing and auditing the skill itself |
|
|
60
|
+
| [Adoption](docs/adoption.md) | Wiring the lint into another repo |
|
|
61
|
+
| [Releasing](docs/releasing.md) | The npm pipeline |
|
|
62
|
+
| [Contributing](CONTRIBUTING.md) | Local setup and the verify gate |
|
|
63
|
+
| [Security](https://github.com/uinaf/skillcheck/security/policy) | Reporting a vulnerability |
|
|
64
|
+
|
|
65
|
+
## License
|
|
63
66
|
|
|
64
67
|
MIT · undefined is not a function LLC
|
package/dist/cli.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import { lintSkills } from "./lint.js";
|
|
3
|
-
import { generateRun, runNameFor } from "./scenario.js";
|
|
3
|
+
import { generateRun, requiredEvalPackages, resolvePackageDir, runNameFor } from "./scenario.js";
|
|
4
4
|
import { execFileSync, spawnSync } from "node:child_process";
|
|
5
5
|
import fs from "node:fs";
|
|
6
6
|
import path from "node:path";
|
|
@@ -81,12 +81,35 @@ function runOptions(flags) {
|
|
|
81
81
|
maxTurns: flags.has("--max-turns") ? parseMaxTurns(flags.get("--max-turns")) : void 0
|
|
82
82
|
};
|
|
83
83
|
}
|
|
84
|
+
function ensureEvalPackages(opts) {
|
|
85
|
+
const missing = requiredEvalPackages(opts, process.env.ANTHROPIC_API_KEY !== void 0).filter((pkg) => resolvePackageDir(pkg) === void 0);
|
|
86
|
+
if (missing.length === 0) return;
|
|
87
|
+
const peers = JSON.parse(fs.readFileSync(path.join(packageDir, "package.json"), "utf8")).peerDependencies ?? {};
|
|
88
|
+
const specs = missing.map((pkg) => `"${pkg}@${peers[pkg] ?? "latest"}"`).join(" ");
|
|
89
|
+
console.error([
|
|
90
|
+
`missing eval package(s): ${missing.join(", ")}`,
|
|
91
|
+
"",
|
|
92
|
+
"The eval engine is an optional peer so `skillcheck lint` installs stay",
|
|
93
|
+
"small. Evals are operator-run; install the peers next to @uinaf/skillcheck:",
|
|
94
|
+
"",
|
|
95
|
+
` pnpm add -D ${specs}`
|
|
96
|
+
].join("\n"));
|
|
97
|
+
process.exit(2);
|
|
98
|
+
}
|
|
99
|
+
function promptfooEntry() {
|
|
100
|
+
const dir = resolvePackageDir("promptfoo");
|
|
101
|
+
if (dir === void 0) throw new Error("promptfoo is not installed");
|
|
102
|
+
const bin = JSON.parse(fs.readFileSync(path.join(dir, "package.json"), "utf8")).bin;
|
|
103
|
+
const rel = typeof bin === "string" ? bin : bin?.promptfoo;
|
|
104
|
+
if (typeof rel !== "string") throw new Error(`promptfoo at ${dir} declares no bin`);
|
|
105
|
+
return path.join(dir, rel);
|
|
106
|
+
}
|
|
84
107
|
function classifyResult(raw) {
|
|
85
108
|
const root = raw;
|
|
86
109
|
const res = root?.results?.results?.[0];
|
|
87
|
-
if (res === void 0) return { error: "promptfoo output carried no result" };
|
|
110
|
+
if (res === void 0 || res === null) return { error: "promptfoo output carried no result" };
|
|
88
111
|
const message = typeof res.error === "string" ? res.error.trim() : "";
|
|
89
|
-
const stats = root?.results?.stats;
|
|
112
|
+
const stats = root?.results?.stats ?? void 0;
|
|
90
113
|
const gradedByStats = stats !== void 0 && (stats.errors ?? 0) === 0 && ((stats.failures ?? 0) > 0 || (stats.successes ?? 0) > 0);
|
|
91
114
|
if (message !== "" && !gradedByStats) return { error: message };
|
|
92
115
|
if (stats !== void 0 && (stats.errors ?? 0) > 0 && (stats.successes ?? 0) === 0 && (stats.failures ?? 0) === 0) return { error: "promptfoo reported an errored test with nothing graded" };
|
|
@@ -105,6 +128,9 @@ function gitHead(root) {
|
|
|
105
128
|
function metaPath(resultPath) {
|
|
106
129
|
return resultPath.replace(/\.json$/, ".meta.json");
|
|
107
130
|
}
|
|
131
|
+
function attemptPath(resultPath) {
|
|
132
|
+
return `${resultPath}.attempt`;
|
|
133
|
+
}
|
|
108
134
|
function runScenario(scenarioDir, opts, root) {
|
|
109
135
|
const dirs = stateDirs(root);
|
|
110
136
|
const { name, configPath } = generateRun(path.resolve(scenarioDir), opts, {
|
|
@@ -114,11 +140,12 @@ function runScenario(scenarioDir, opts, root) {
|
|
|
114
140
|
});
|
|
115
141
|
fs.mkdirSync(dirs.results, { recursive: true });
|
|
116
142
|
const resultPath = path.join(dirs.results, `${name}.json`);
|
|
143
|
+
fs.writeFileSync(attemptPath(resultPath), "{}\n");
|
|
117
144
|
fs.rmSync(resultPath, { force: true });
|
|
118
145
|
fs.rmSync(metaPath(resultPath), { force: true });
|
|
119
146
|
const sha = gitHead(root);
|
|
120
|
-
const rc = spawnSync(
|
|
121
|
-
|
|
147
|
+
const rc = spawnSync(process.execPath, [
|
|
148
|
+
promptfooEntry(),
|
|
122
149
|
"eval",
|
|
123
150
|
"--no-cache",
|
|
124
151
|
"--no-progress-bar",
|
|
@@ -151,12 +178,15 @@ function runScenario(scenarioDir, opts, root) {
|
|
|
151
178
|
outcome.score = verdict.score;
|
|
152
179
|
outcome.pass = verdict.pass;
|
|
153
180
|
outcome.error = verdict.error;
|
|
154
|
-
if (verdict.score !== void 0)
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
181
|
+
if (verdict.score !== void 0) {
|
|
182
|
+
fs.writeFileSync(metaPath(resultPath), JSON.stringify({
|
|
183
|
+
skills_tree_sha: sha,
|
|
184
|
+
harness: opts.harness,
|
|
185
|
+
ran_at: (/* @__PURE__ */ new Date()).toISOString(),
|
|
186
|
+
tool_version: toolVersion()
|
|
187
|
+
}, null, 2) + "\n");
|
|
188
|
+
fs.rmSync(attemptPath(resultPath), { force: true });
|
|
189
|
+
}
|
|
160
190
|
return outcome;
|
|
161
191
|
}
|
|
162
192
|
function discoverScenarios(root) {
|
|
@@ -182,7 +212,9 @@ function discoverScenarios(root) {
|
|
|
182
212
|
function cmdRun(argv) {
|
|
183
213
|
const { positional, flags } = parseArgs(argv);
|
|
184
214
|
if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--judge-effort EFFORT] [--harness claude|codex|cursor]");
|
|
185
|
-
const
|
|
215
|
+
const opts = runOptions(flags);
|
|
216
|
+
ensureEvalPackages(opts);
|
|
217
|
+
const o = runScenario(positional[0], opts, resolveRoot(flags));
|
|
186
218
|
if (o.score === void 0) {
|
|
187
219
|
console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
|
|
188
220
|
process.exit(2);
|
|
@@ -195,6 +227,7 @@ function cmdSweep(argv) {
|
|
|
195
227
|
if (positional.length > 0) fail("usage: skillcheck sweep [--root DIR] [--all]");
|
|
196
228
|
const root = resolveRoot(flags);
|
|
197
229
|
const opts = runOptions(flags);
|
|
230
|
+
ensureEvalPackages(opts);
|
|
198
231
|
const all = flags.get("--all") === true;
|
|
199
232
|
const resultsDir = stateDirs(root).results;
|
|
200
233
|
let passed = 0, failed = 0, errored = 0, skipped = 0;
|
|
@@ -221,11 +254,30 @@ function cmdSweep(argv) {
|
|
|
221
254
|
console.log(`\nsweep: ${passed} passed, ${failed} failed, ${errored} errored, ${skipped} skipped`);
|
|
222
255
|
process.exit(errored > 0 ? 2 : failed > 0 ? 1 : 0);
|
|
223
256
|
}
|
|
257
|
+
function resultIdentity(file) {
|
|
258
|
+
const base = file.replace(/\.json$/, "");
|
|
259
|
+
const suffix = base.match(/--(codex|cursor)$/);
|
|
260
|
+
const harness = suffix === null ? "claude" : suffix[1];
|
|
261
|
+
const [skill, ...rest] = base.replace(/--(codex|cursor)$/, "").split("--");
|
|
262
|
+
return {
|
|
263
|
+
skill,
|
|
264
|
+
scenario: rest.join("--"),
|
|
265
|
+
harness
|
|
266
|
+
};
|
|
267
|
+
}
|
|
224
268
|
function reduceResults(dir, allowMixed) {
|
|
225
269
|
const entries = [];
|
|
226
270
|
const skipped = [];
|
|
227
271
|
const shas = /* @__PURE__ */ new Set();
|
|
228
|
-
|
|
272
|
+
const files = fs.readdirSync(dir);
|
|
273
|
+
const incomplete = new Set(files.filter((f) => f.endsWith(".json.attempt")).map((f) => f.replace(/\.attempt$/, "")));
|
|
274
|
+
const results = files.filter((f) => f.endsWith(".json") && !f.endsWith(".meta.json"));
|
|
275
|
+
for (const f of [.../* @__PURE__ */ new Set([...results, ...incomplete])].sort()) {
|
|
276
|
+
if (incomplete.has(f)) {
|
|
277
|
+
console.error(`skipping ${f}: attempt did not complete with a graded result`);
|
|
278
|
+
skipped.push(f);
|
|
279
|
+
continue;
|
|
280
|
+
}
|
|
229
281
|
let raw;
|
|
230
282
|
try {
|
|
231
283
|
raw = JSON.parse(fs.readFileSync(path.join(dir, f), "utf8"));
|
|
@@ -233,17 +285,16 @@ function reduceResults(dir, allowMixed) {
|
|
|
233
285
|
raw = void 0;
|
|
234
286
|
}
|
|
235
287
|
const res = raw?.results?.results?.[0];
|
|
236
|
-
|
|
237
|
-
|
|
288
|
+
const verdict = classifyResult(raw);
|
|
289
|
+
if (verdict.score === void 0 || verdict.pass === void 0) {
|
|
290
|
+
console.error(`skipping ${f}: ${verdict.error}`);
|
|
238
291
|
skipped.push(f);
|
|
239
292
|
continue;
|
|
240
293
|
}
|
|
241
294
|
const provider = raw.config?.providers?.[0];
|
|
242
295
|
const judge = raw.config?.defaultTest?.options?.provider;
|
|
243
296
|
const base = f.replace(/\.json$/, "");
|
|
244
|
-
const
|
|
245
|
-
const harness = suffix === null ? "claude" : suffix[1];
|
|
246
|
-
const [skill, ...rest] = base.replace(/--(codex|cursor)$/, "").split("--");
|
|
297
|
+
const { skill, scenario, harness } = resultIdentity(f);
|
|
247
298
|
let sha = "unattested";
|
|
248
299
|
try {
|
|
249
300
|
sha = JSON.parse(fs.readFileSync(path.join(dir, `${base}.meta.json`), "utf8")).skills_tree_sha ?? "unattested";
|
|
@@ -251,11 +302,11 @@ function reduceResults(dir, allowMixed) {
|
|
|
251
302
|
shas.add(sha);
|
|
252
303
|
entries.push({
|
|
253
304
|
skill,
|
|
254
|
-
scenario
|
|
305
|
+
scenario,
|
|
255
306
|
harness,
|
|
256
307
|
skills_tree_sha: sha,
|
|
257
|
-
score:
|
|
258
|
-
pass:
|
|
308
|
+
score: verdict.score,
|
|
309
|
+
pass: verdict.pass,
|
|
259
310
|
agent_model: provider?.config?.model ?? `${harness}-default`,
|
|
260
311
|
judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? judge?.id ?? "unknown",
|
|
261
312
|
latency_ms: res.latencyMs,
|
|
@@ -309,15 +360,19 @@ function cmdSummarize(argv) {
|
|
|
309
360
|
const { positional, flags } = parseArgs(argv);
|
|
310
361
|
if (positional.length > 0) fail("usage: skillcheck summarize [--root DIR] [--allow-mixed]");
|
|
311
362
|
const dirs = stateDirs(resolveRoot(flags));
|
|
312
|
-
if (!fs.existsSync(dirs.results)) fail(`no results directory at ${dirs.results}
|
|
363
|
+
if (!fs.existsSync(dirs.results)) fail(`no results directory at ${dirs.results}; run some evals first`);
|
|
313
364
|
const { entries, skipped } = reduceResults(dirs.results, flags.get("--allow-mixed") === true);
|
|
314
365
|
fs.mkdirSync(dirs.scorecards, { recursive: true });
|
|
315
366
|
const out = path.join(dirs.scorecards, `${(/* @__PURE__ */ new Date()).toISOString().slice(0, 10)}.json`);
|
|
316
367
|
const existing = readExistingScorecard(out);
|
|
368
|
+
const skippedKeys = new Set(skipped.map((file) => entryKey(resultIdentity(file))));
|
|
369
|
+
if (existing.some((entry) => skippedKeys.has(entryKey(entry)))) throw new Error("skipped rerun matches an existing score; refusing to carry it or overwrite the scorecard");
|
|
317
370
|
const merged = mergeScorecard(existing, entries);
|
|
371
|
+
const treeSha = treeShaOf(merged.entries);
|
|
372
|
+
if (treeSha === "mixed" && flags.get("--allow-mixed") !== true) throw new Error("scorecard spans multiple skills-tree revisions; rerun stale ones or pass --allow-mixed");
|
|
318
373
|
const scorecard = {
|
|
319
374
|
ran_at: (/* @__PURE__ */ new Date()).toISOString(),
|
|
320
|
-
skills_tree_sha:
|
|
375
|
+
skills_tree_sha: treeSha,
|
|
321
376
|
scenarios: merged.entries
|
|
322
377
|
};
|
|
323
378
|
fs.writeFileSync(out, JSON.stringify(scorecard, null, 2) + "\n");
|
package/dist/scenario.js
CHANGED
|
@@ -213,8 +213,25 @@ function buildConfig(s, workdir, manifestPath, opts, paths) {
|
|
|
213
213
|
}
|
|
214
214
|
const SDK_PACKAGES = ["@anthropic-ai/claude-agent-sdk", "@openai/codex-sdk"];
|
|
215
215
|
function sdkNodeModulesDir() {
|
|
216
|
+
for (const pkg of SDK_PACKAGES) {
|
|
217
|
+
const dir = holdingNodeModules(pkg);
|
|
218
|
+
if (dir !== void 0) return dir;
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
function holdingNodeModules(pkg) {
|
|
216
222
|
const require = createRequire(import.meta.url);
|
|
217
|
-
for (const
|
|
223
|
+
for (const dir of require.resolve.paths(pkg) ?? []) if (fs.existsSync(path.join(dir, pkg, "package.json"))) return dir;
|
|
224
|
+
}
|
|
225
|
+
function resolvePackageDir(pkg) {
|
|
226
|
+
const dir = holdingNodeModules(pkg);
|
|
227
|
+
return dir === void 0 ? void 0 : path.join(dir, pkg);
|
|
228
|
+
}
|
|
229
|
+
function requiredEvalPackages(opts, hasAnthropicKey) {
|
|
230
|
+
const pkgs = ["promptfoo"];
|
|
231
|
+
const sdkJudge = !opts.judgeModel.includes(":") && !hasAnthropicKey;
|
|
232
|
+
if (opts.harness === "claude" || sdkJudge) pkgs.push("@anthropic-ai/claude-agent-sdk");
|
|
233
|
+
if (opts.harness === "codex") pkgs.push("@openai/codex-sdk");
|
|
234
|
+
return pkgs;
|
|
218
235
|
}
|
|
219
236
|
function generateRun(scenarioDir, opts, paths) {
|
|
220
237
|
const s = loadScenario(scenarioDir);
|
|
@@ -239,4 +256,4 @@ function generateRun(scenarioDir, opts, paths) {
|
|
|
239
256
|
};
|
|
240
257
|
}
|
|
241
258
|
//#endregion
|
|
242
|
-
export { buildConfig, generateRun, isHiddenSkill, loadScenario, materialize, runNameFor, sdkNodeModulesDir, stripHiddenFlag };
|
|
259
|
+
export { buildConfig, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag };
|
package/docs/adoption.md
CHANGED
|
@@ -1,15 +1,15 @@
|
|
|
1
|
-
#
|
|
1
|
+
# Adopting it in a repo
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
model auth.
|
|
3
|
+
Two separate decisions: run the lint in CI, and run evals on a machine that has
|
|
4
|
+
model auth. Only the first belongs in a consumer repo.
|
|
5
5
|
|
|
6
|
-
##
|
|
6
|
+
## Lint in CI
|
|
7
7
|
|
|
8
8
|
```sh
|
|
9
9
|
pnpm add -D @uinaf/skillcheck
|
|
10
10
|
```
|
|
11
11
|
|
|
12
|
-
|
|
12
|
+
Runners that install with `--ignore-scripts` are fine: the package ships
|
|
13
13
|
compiled ESM and has no install, prepare, or postinstall script.
|
|
14
14
|
|
|
15
15
|
```json
|
|
@@ -32,49 +32,59 @@ jobs:
|
|
|
32
32
|
- run: pnpm run skills:lint
|
|
33
33
|
```
|
|
34
34
|
|
|
35
|
-
|
|
35
|
+
The job needs no secrets and no network beyond the install. Node 24 is the
|
|
36
36
|
floor.
|
|
37
37
|
|
|
38
|
-
|
|
38
|
+
Run it through the script rather than a bare `npx skillcheck`: the script
|
|
39
39
|
resolves the version the repo pinned, and `npx` would resolve the latest one on
|
|
40
40
|
the registry.
|
|
41
41
|
|
|
42
|
-
##
|
|
42
|
+
## Evals
|
|
43
43
|
|
|
44
|
-
|
|
45
|
-
operator machine or a job that already holds gateway auth
|
|
44
|
+
Sweeps need model credentials, so they stay off consumer CI and run from an
|
|
45
|
+
operator machine or a job that already holds gateway auth. They also need the
|
|
46
|
+
eval engine, which is an optional peer precisely so the lint-only install
|
|
47
|
+
above stays small. Install it next to the package on the operator machine:
|
|
48
|
+
|
|
49
|
+
```sh
|
|
50
|
+
pnpm add -D promptfoo @anthropic-ai/claude-agent-sdk @openai/codex-sdk
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
`run` and `sweep` check for the peers the selected harness needs and exit 2
|
|
54
|
+
with that install command when they are missing, so a lint-only install never
|
|
55
|
+
crashes into a resolution error. Then:
|
|
46
56
|
|
|
47
57
|
```sh
|
|
48
58
|
skillcheck sweep # resumes: only scenarios without results
|
|
49
59
|
skillcheck summarize # writes .skillcheck/scorecards/<UTC-date>.json
|
|
50
60
|
```
|
|
51
61
|
|
|
52
|
-
|
|
62
|
+
Commit `.skillcheck/scorecards/`. Gitignore the rest:
|
|
53
63
|
|
|
54
64
|
```gitignore
|
|
55
65
|
.skillcheck/results/
|
|
56
66
|
.skillcheck/scratch/
|
|
57
67
|
```
|
|
58
68
|
|
|
59
|
-
|
|
69
|
+
A scorecard is only comparable against the tree it graded, which is why every
|
|
60
70
|
result carries the root repo's HEAD and `summarize` refuses to mix revisions
|
|
61
71
|
without `--allow-mixed`.
|
|
62
72
|
|
|
63
|
-
##
|
|
73
|
+
## Upgrading
|
|
64
74
|
|
|
65
|
-
|
|
75
|
+
Bump the version in `package.json` and rerun the sweep. Results carry the
|
|
66
76
|
`tool_version` that produced them, so a scorecard says which harness build it
|
|
67
77
|
came from as well as which skills tree.
|
|
68
78
|
|
|
69
|
-
##
|
|
79
|
+
## The older install path
|
|
70
80
|
|
|
71
|
-
|
|
81
|
+
Before the package was published, consumers installed it from a git tag:
|
|
72
82
|
|
|
73
83
|
```sh
|
|
74
84
|
npm i -D github:uinaf/skillcheck#v0.1.3
|
|
75
85
|
```
|
|
76
86
|
|
|
77
|
-
|
|
78
|
-
12 consumers needed `allow-git=root` in `.npmrc` to accept the spec at all.
|
|
79
|
-
registry install needs none of that.
|
|
87
|
+
Those tags are frozen and still work: they carry a committed `dist/`, and npm
|
|
88
|
+
12 consumers needed `allow-git=root` in `.npmrc` to accept the spec at all. The
|
|
89
|
+
registry install needs none of that. Tags from `v0.1.4` on are npm releases and
|
|
80
90
|
carry no `dist/`, so a git spec pointing at one will not run.
|
package/docs/authoring.md
CHANGED
|
@@ -1,70 +1,70 @@
|
|
|
1
|
-
#
|
|
1
|
+
# Authoring and auditing skills
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
cheap to load.
|
|
3
|
+
What `lint` and evals cannot judge: whether a skill is worth routing to and
|
|
4
|
+
cheap to load. Use this when writing a skill or auditing one. Evidence beats
|
|
5
5
|
stylistic preference; run `skillcheck lint` first and let this cover the rest.
|
|
6
6
|
|
|
7
|
-
##
|
|
7
|
+
## Metadata and discovery
|
|
8
8
|
|
|
9
9
|
- `name` is concrete and easy to say out loud. `helper`, `tools`, `utils` are
|
|
10
10
|
discovery smells.
|
|
11
11
|
- `description` is third person and says both what the skill does and when to
|
|
12
|
-
use it.
|
|
12
|
+
use it. It is an always-loaded retrieval pointer: front-load the concrete
|
|
13
13
|
action or domain that should activate it.
|
|
14
|
-
-
|
|
14
|
+
- One trigger per materially distinct request branch. Collapse synonyms that
|
|
15
15
|
rename the same branch.
|
|
16
|
-
-
|
|
16
|
+
- State the main overlap boundary without naming another skill.
|
|
17
17
|
|
|
18
|
-
##
|
|
18
|
+
## Body shape
|
|
19
19
|
|
|
20
|
-
-
|
|
20
|
+
- Keep `SKILL.md` on workflow, principles, boundaries, and routing. Lead with
|
|
21
21
|
the task, not a bibliography.
|
|
22
|
-
-
|
|
22
|
+
- Assume the model is smart; spend tokens on repo-specific judgment. Delete any
|
|
23
23
|
instruction that would not change a capable model's behavior.
|
|
24
|
-
-
|
|
24
|
+
- Match freedom to risk: high for contextual judgment, medium when a preferred
|
|
25
25
|
pattern exists, low for fragile operations.
|
|
26
|
-
-
|
|
26
|
+
- Say what evidence to gather and what a complete result includes. End each
|
|
27
27
|
step with an observable completion condition, not "understood" or "handled".
|
|
28
28
|
|
|
29
|
-
##
|
|
29
|
+
## Progressive disclosure
|
|
30
30
|
|
|
31
|
-
-
|
|
32
|
-
`SKILL.md`, each with a task-shaped retrieval job.
|
|
31
|
+
- Durable detail, rubrics, and long examples go in `references/`, one hop from
|
|
32
|
+
`SKILL.md`, each with a task-shaped retrieval job. Material every path needs
|
|
33
33
|
stays inline.
|
|
34
|
-
-
|
|
34
|
+
- For repeated deterministic work, route to the target's existing framework,
|
|
35
35
|
schema, task graph, or library; otherwise add a tested module in the
|
|
36
36
|
project's primary language, not ad-hoc shell rendered as prose.
|
|
37
|
-
-
|
|
37
|
+
- When executable code belongs to another maintained project, link the exact
|
|
38
38
|
public artifact and state the contract it demonstrates; do not fork it into
|
|
39
39
|
prose.
|
|
40
|
-
-
|
|
41
|
-
next steps locally.
|
|
40
|
+
- A package stays independently usable: state prerequisites and out-of-scope
|
|
41
|
+
next steps locally. Never invoke, import, or assume a sibling skill.
|
|
42
42
|
|
|
43
|
-
##
|
|
43
|
+
## Audit
|
|
44
44
|
|
|
45
|
-
|
|
45
|
+
Grade each dimension strong, mixed, or weak:
|
|
46
46
|
|
|
47
|
-
|
|
|
47
|
+
| Dimension | Question |
|
|
48
48
|
| ---------------------- | --------------------------------------------------------------------- |
|
|
49
|
-
|
|
|
50
|
-
|
|
|
51
|
-
|
|
|
52
|
-
|
|
|
53
|
-
|
|
|
54
|
-
|
|
|
49
|
+
| Discovery | Does metadata alone route a realistic request here |
|
|
50
|
+
| Workflow | Does the body say how to begin, what evidence to gather, when to stop |
|
|
51
|
+
| Progressive disclosure | Is detail in the right file |
|
|
52
|
+
| Repo fit | Are links, commands, and conventions current |
|
|
53
|
+
| Verification | Is the strongest mechanical check named, plus a real evidence loop |
|
|
54
|
+
| Boundaries | Are limits and next steps stated without leaning on a sibling skill |
|
|
55
55
|
|
|
56
|
-
|
|
56
|
+
Blockers, must-fix: invalid frontmatter; a description that fails discovery;
|
|
57
57
|
stale commands, paths, or links; a workflow with no start, evidence loop, or
|
|
58
58
|
completion; conflicts with the repo's guidance; sibling-skill dependencies.
|
|
59
59
|
|
|
60
|
-
|
|
60
|
+
Major findings: vague name; synonym-stuffed description; bloated `SKILL.md`;
|
|
61
61
|
missing or muddy boundaries; prose re-inventing a deterministic tool; abstract
|
|
62
62
|
examples.
|
|
63
63
|
|
|
64
|
-
##
|
|
64
|
+
## Improve
|
|
65
65
|
|
|
66
|
-
|
|
67
|
-
change that improves activation, decision quality, or proof.
|
|
66
|
+
Fix blockers first, then the highest-leverage majors. Prefer the smallest
|
|
67
|
+
change that improves activation, decision quality, or proof. When pruning,
|
|
68
68
|
measure common-path context for representative requests; line count alone does
|
|
69
|
-
not reveal retrieval cost.
|
|
69
|
+
not reveal retrieval cost. After edits, rerun `skillcheck lint` and the repo's
|
|
70
70
|
gate, and rerun evals when behavior was the thing changed.
|
package/docs/releasing.md
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
|
-
#
|
|
1
|
+
# Releasing
|
|
2
2
|
|
|
3
|
-
##
|
|
3
|
+
## Pipeline
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
A push to `main` runs one workflow, `.github/workflows/release.yml`:
|
|
6
6
|
|
|
7
7
|
```text
|
|
8
8
|
verify ──┐
|
|
@@ -12,69 +12,69 @@ scan ────┘
|
|
|
12
12
|
|
|
13
13
|
`verify` and `scan` are the shared gate: `verify` is called from `verify.yml`,
|
|
14
14
|
and `scan` calls the shared scan in `uinaf/.github`, the same one `scan.yml`
|
|
15
|
-
runs for pull requests.
|
|
15
|
+
runs for pull requests. Keep it that way: a second copy of the gate on a push-to-`main` workflow
|
|
16
16
|
races this one over the same commit.
|
|
17
17
|
|
|
18
|
-
|
|
18
|
+
The file name `release.yml` is load-bearing. See below.
|
|
19
19
|
|
|
20
20
|
## npm
|
|
21
21
|
|
|
22
22
|
`@uinaf/skillcheck` publishes from `.github/workflows/release.yml` via npm
|
|
23
|
-
Trusted Publishing (OIDC) and the `uinaf-releaser` GitHub App.
|
|
23
|
+
Trusted Publishing (OIDC) and the `uinaf-releaser` GitHub App. There is no npm
|
|
24
24
|
token in this repository, in its environments, or in the organization.
|
|
25
25
|
|
|
26
|
-
|
|
26
|
+
Required on the `release` GitHub Environment:
|
|
27
27
|
|
|
28
|
-
|
|
|
28
|
+
| Name | Kind | Purpose |
|
|
29
29
|
| ------------------------------- | ------ | ------------------------------------------- |
|
|
30
30
|
| `UINAF_RELEASE_APP_CLIENT_ID` | var | GitHub App client id for the releaser bot |
|
|
31
31
|
| `UINAF_RELEASE_APP_PRIVATE_KEY` | secret | GitHub App private key for the releaser bot |
|
|
32
32
|
|
|
33
|
-
|
|
33
|
+
The trusted publisher on npmjs.com is registered by **file path**, so
|
|
34
34
|
`.github/workflows/release.yml` cannot be renamed or moved without editing that
|
|
35
|
-
registration first.
|
|
36
|
-
nothing earlier in the run reports it.
|
|
35
|
+
registration first. A rename fails the publish with an identity mismatch, and
|
|
36
|
+
nothing earlier in the run reports it. The `release` environment name is bound
|
|
37
37
|
the same way.
|
|
38
38
|
|
|
39
|
-
|
|
39
|
+
Deleting the `release` environment deletes both rows above with it, and there is
|
|
40
40
|
no repo-level fallback: `create-github-app-token` then runs with empty inputs
|
|
41
|
-
and the job fails at that step.
|
|
41
|
+
and the job fails at that step. The private key cannot be read back from
|
|
42
42
|
GitHub; recreating it means generating a new one in the App settings.
|
|
43
43
|
|
|
44
|
-
##
|
|
44
|
+
## Version history
|
|
45
45
|
|
|
46
|
-
|
|
46
|
+
The version and the tag are owned by semantic-release. `tagFormat` is `v${version}` and
|
|
47
47
|
history continues from `v0.1.3`; `v0.1.0`–`v0.1.3` are the legacy git-install
|
|
48
48
|
tags and are never deleted or moved.
|
|
49
49
|
|
|
50
|
-
|
|
50
|
+
During preparation, `@semantic-release/npm` stages the released `package.json`
|
|
51
51
|
version and `@jno21/semantic-release-github-commit` commits it to `main` through
|
|
52
52
|
GitHub's API as the authenticated App. GitHub signs that commit, and the release
|
|
53
|
-
tag points to it.
|
|
53
|
+
tag points to it. The `[skip ci]` marker on that commit is what stops a release
|
|
54
54
|
from releasing itself.
|
|
55
55
|
|
|
56
|
-
|
|
56
|
+
Check what the next version would be without publishing anything:
|
|
57
57
|
|
|
58
58
|
```sh
|
|
59
59
|
pnpm dlx semantic-release --dry-run --no-ci
|
|
60
60
|
```
|
|
61
61
|
|
|
62
|
-
##
|
|
62
|
+
## The artifact
|
|
63
63
|
|
|
64
64
|
`dist/` is generated and untracked. `prepublishOnly` runs `pnpm run verify`,
|
|
65
65
|
which builds it, so the tarball is always packed from a tree that just passed
|
|
66
66
|
the gate. `files` is `dist`, `docs`, `README.md`, `LICENSE`.
|
|
67
67
|
|
|
68
|
-
|
|
68
|
+
Manual publish is emergency recovery only:
|
|
69
69
|
|
|
70
70
|
```sh
|
|
71
71
|
pnpm run verify
|
|
72
72
|
npm publish --access public
|
|
73
73
|
```
|
|
74
74
|
|
|
75
|
-
##
|
|
75
|
+
## The old install path
|
|
76
76
|
|
|
77
|
-
|
|
78
|
-
`github:uinaf/skillcheck#v0.1.3`.
|
|
79
|
-
committed `dist/`, so anything pinned to them keeps working untouched.
|
|
77
|
+
Before `@uinaf/skillcheck` existed, consumers installed
|
|
78
|
+
`github:uinaf/skillcheck#v0.1.3`. Those tags still resolve and still carry a
|
|
79
|
+
committed `dist/`, so anything pinned to them keeps working untouched. New
|
|
80
80
|
consumers use npm.
|
package/docs/scenarios.md
CHANGED
|
@@ -1,20 +1,20 @@
|
|
|
1
|
-
#
|
|
1
|
+
# Writing scenarios
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
A scenario is two files in a frozen location:
|
|
4
4
|
|
|
5
5
|
```text
|
|
6
6
|
<root>/skills/<skill>/evals/<scenario>/task.md
|
|
7
7
|
<root>/skills/<skill>/evals/<scenario>/criteria.json
|
|
8
8
|
```
|
|
9
9
|
|
|
10
|
-
|
|
11
|
-
and the scorecard entry.
|
|
10
|
+
The path is the identity: `<skill>--<scenario>` names the run, the result file,
|
|
11
|
+
and the scorecard entry. On the codex and cursor harnesses the name gains a
|
|
12
12
|
`--codex` or `--cursor` suffix, so every harness can hold results side by side.
|
|
13
|
-
|
|
13
|
+
A directory missing either file is not discovered.
|
|
14
14
|
|
|
15
15
|
## task.md
|
|
16
16
|
|
|
17
|
-
|
|
17
|
+
The prompt handed to the agent, verbatim, with one piece of syntax. Input files
|
|
18
18
|
are embedded inline and materialized into the workdir before the run:
|
|
19
19
|
|
|
20
20
|
```md
|
|
@@ -25,13 +25,13 @@ Fix the failing check in the config below.
|
|
|
25
25
|
======= END FILE =======
|
|
26
26
|
```
|
|
27
27
|
|
|
28
|
-
|
|
29
|
-
is available in your working directory.") and written to disk.
|
|
28
|
+
Each block is replaced in the prompt with a pointer ("Input file `config.json`
|
|
29
|
+
is available in your working directory.") and written to disk. Destinations
|
|
30
30
|
must stay under the workdir, must not collide, and must not target `.claude/`,
|
|
31
31
|
`.agents/`, or `.cursor/`, since a fixture that writes agent config would be
|
|
32
32
|
configuring its own examiner.
|
|
33
33
|
|
|
34
|
-
|
|
34
|
+
Write the task the way a user would write it. Do not name the skill, describe
|
|
35
35
|
its steps, or hint at the checklist: routing is part of what is being measured.
|
|
36
36
|
|
|
37
37
|
## criteria.json
|
|
@@ -46,45 +46,45 @@ its steps, or hint at the checklist: routing is part of what is being measured.
|
|
|
46
46
|
}
|
|
47
47
|
```
|
|
48
48
|
|
|
49
|
-
`type` must be `weighted_checklist` and the checklist must be non-empty.
|
|
49
|
+
`type` must be `weighted_checklist` and the checklist must be non-empty. Every
|
|
50
50
|
item needs a non-empty `name` and `description` and a positive `max_score`.
|
|
51
51
|
|
|
52
|
-
|
|
53
|
-
assert-set with threshold 0.7.
|
|
52
|
+
Each item becomes one `llm-rubric` assertion weighted by `max_score`, inside an
|
|
53
|
+
assert-set with threshold 0.7. A separate `skill-used` assertion sits outside
|
|
54
54
|
that aggregate, so a run that produces good output without ever loading the
|
|
55
|
-
skill still fails.
|
|
55
|
+
skill still fails. There is no test-level threshold: both must pass.
|
|
56
56
|
|
|
57
|
-
|
|
58
|
-
property, not a feeling.
|
|
57
|
+
Write descriptions a judge can check against the deliverable: an observable
|
|
58
|
+
property, not a feeling. Weight the items that would make a reviewer reject the
|
|
59
59
|
work.
|
|
60
60
|
|
|
61
|
-
##
|
|
61
|
+
## What the judge sees
|
|
62
62
|
|
|
63
|
-
|
|
64
|
-
pre-run manifest.
|
|
63
|
+
The agent's final message, plus every file in the workdir that differs from the
|
|
64
|
+
pre-run manifest. Unchanged inputs are omitted; deleted inputs, unreadable
|
|
65
65
|
files, and non-regular files are named rather than read.
|
|
66
66
|
|
|
67
|
-
|
|
68
|
-
appended total at 24,000, with truncation stated inline.
|
|
69
|
-
rubric judges return nothing at all, which is why the caps exist.
|
|
67
|
+
Sections are sorted by path, each file is capped at 4,000 characters and the
|
|
68
|
+
appended total at 24,000, with truncation stated inline. Very large outputs make
|
|
69
|
+
rubric judges return nothing at all, which is why the caps exist. Keep fixtures
|
|
70
70
|
small enough that the deliverable fits.
|
|
71
71
|
|
|
72
|
-
##
|
|
72
|
+
## Hidden skills
|
|
73
73
|
|
|
74
|
-
|
|
75
|
-
production, which the agent SDK cannot simulate.
|
|
74
|
+
A skill with `disable-model-invocation: true` is explicit-invoke-only in
|
|
75
|
+
production, which the agent SDK cannot simulate. So the eval copy, never the shipped one,
|
|
76
76
|
has the flag stripped, and the task gains a leading
|
|
77
|
-
`Use the <skill> skill for this task.`
|
|
78
|
-
behavior-when-invoked rather than routing.
|
|
77
|
+
`Use the <skill> skill for this task.` The eval then measures
|
|
78
|
+
behavior-when-invoked rather than routing. The flag is only honored inside the
|
|
79
79
|
frontmatter block; body text mentioning the key does not count.
|
|
80
80
|
|
|
81
|
-
##
|
|
81
|
+
## The workdir
|
|
82
82
|
|
|
83
|
-
|
|
84
|
-
time.
|
|
83
|
+
Per run, under `<root>/.skillcheck/scratch/<name>/`, rebuilt from scratch each
|
|
84
|
+
time. The skill under test is installed where the harness discovers skills:
|
|
85
85
|
`.claude/skills/<skill>/`, plus `.agents/skills/<skill>/` on codex, or
|
|
86
|
-
`.cursor/skills/<skill>/` alone on cursor
|
|
86
|
+
`.cursor/skills/<skill>/` alone on cursor, with its `evals/` directory
|
|
87
87
|
excluded, so criteria never leak into the agent's context.
|
|
88
88
|
|
|
89
|
-
|
|
89
|
+
Scenario quality is behavioral proof; [authoring](authoring.md) covers the
|
|
90
90
|
judgment layer lint and evals cannot grade.
|
package/docs/usage.md
CHANGED
|
@@ -1,60 +1,62 @@
|
|
|
1
|
-
#
|
|
1
|
+
# Usage
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Every subcommand resolves one root, `--root <dir>` or the current directory.
|
|
4
4
|
`lint` also takes the root as a positional, because that is the shape CI reaches
|
|
5
5
|
for first.
|
|
6
6
|
|
|
7
|
-
##
|
|
7
|
+
## Lint
|
|
8
8
|
|
|
9
9
|
```sh
|
|
10
10
|
skillcheck lint # lints the current repo
|
|
11
11
|
skillcheck lint ../other # lints another root
|
|
12
12
|
```
|
|
13
13
|
|
|
14
|
-
|
|
14
|
+
Checks each `<root>/skills/<skill>/`:
|
|
15
15
|
|
|
16
|
-
-
|
|
17
|
-
-
|
|
16
|
+
- Frontmatter opens with `---` on line 1 and closes
|
|
17
|
+
- Keys are `name`, `description`, `disable-model-invocation` and nothing else,
|
|
18
18
|
each at most once
|
|
19
19
|
- `name` equals the directory name; `description` is non-empty
|
|
20
20
|
- `disable-model-invocation`, when present, is the bare YAML boolean `true`.
|
|
21
|
-
|
|
22
|
-
-
|
|
21
|
+
A quoted `"true"` is an error
|
|
22
|
+
- Relative links in the body resolve on disk
|
|
23
23
|
|
|
24
|
-
|
|
25
|
-
links never fail.
|
|
24
|
+
Code spans and fenced blocks are stripped before links are checked, so example
|
|
25
|
+
links never fail. External schemes and `#anchors` pass. Dot-directories under
|
|
26
26
|
`skills/` (`.claude-plugin`) are plugin metadata, not packages, and are skipped.
|
|
27
27
|
|
|
28
|
-
|
|
28
|
+
Findings print one per line, relative to the linted root, then a count. Exit 0
|
|
29
29
|
clean, 1 with findings.
|
|
30
30
|
|
|
31
|
-
##
|
|
31
|
+
## Run
|
|
32
32
|
|
|
33
33
|
```sh
|
|
34
34
|
skillcheck run skills/<skill>/evals/<scenario>
|
|
35
35
|
skillcheck run <scenario-dir> --agent MODEL --judge MODEL --harness codex --max-turns 80
|
|
36
36
|
```
|
|
37
37
|
|
|
38
|
-
|
|
38
|
+
Materializes the scenario into `<root>/.skillcheck/scratch/<name>/workdir`,
|
|
39
39
|
installs the skill under test into that workdir, drives the agent, and grades
|
|
40
|
-
the files it wrote.
|
|
41
|
-
usable
|
|
40
|
+
the files it wrote. Exit 0 means pass, 1 means graded fail, and 2 means error.
|
|
41
|
+
Exit 2 covers missing usable promptfoo output or optional eval peers. The
|
|
42
|
+
message carries the exact `pnpm add` command; see
|
|
43
|
+
[adoption](adoption.md#evals).
|
|
42
44
|
|
|
43
|
-
|
|
44
|
-
message, and writes no provenance sidecar.
|
|
45
|
+
A test that errored was never graded, so it exits 2, prints the provider's
|
|
46
|
+
message, and writes no provenance sidecar. It is never reported as
|
|
45
47
|
`FAIL score=0.0000`; only a real judged verdict can fail a run.
|
|
46
48
|
|
|
47
|
-
|
|
48
|
-
`--max-turns 50`.
|
|
49
|
+
Defaults: `--harness claude`, agent `claude-opus-5`, judge `claude-opus-5`,
|
|
50
|
+
`--max-turns 50`. On the codex and cursor harnesses, omitting `--agent` leaves
|
|
49
51
|
the model to that CLI's own default.
|
|
50
52
|
|
|
51
53
|
`--harness cursor` drives the scenario through the Cursor Agent CLI
|
|
52
54
|
(`cursor-agent` on PATH) with the skill installed under `.cursor/skills/`;
|
|
53
|
-
`--agent` names a Cursor model id, e.g. `composer-2.5`.
|
|
55
|
+
`--agent` names a Cursor model id, e.g. `composer-2.5`. There is no promptfoo
|
|
54
56
|
cursor provider, so the run uses this package's own provider module, which
|
|
55
57
|
replays the CLI's `stream-json` output: the `result` event becomes the graded
|
|
56
58
|
output and `SKILL.md` reads under `.cursor/skills/` become the `skill-used`
|
|
57
|
-
evidence.
|
|
59
|
+
evidence. The judge leg is unchanged.
|
|
58
60
|
|
|
59
61
|
`--judge` takes either a bare Claude model (graded through the Anthropic
|
|
60
62
|
selection in [auth](#auth)) or a provider-qualified promptfoo id, passed
|
|
@@ -64,55 +66,63 @@ through verbatim:
|
|
|
64
66
|
skillcheck run <scenario-dir> --judge openai:chat:gpt-5.6-sol --judge-effort high
|
|
65
67
|
```
|
|
66
68
|
|
|
67
|
-
|
|
69
|
+
A provider-qualified judge authenticates through that provider's own env
|
|
68
70
|
(`OPENAI_API_KEY`, plus `OPENAI_BASE_URL` for a gateway) and is recorded
|
|
69
71
|
verbatim in the scorecard's `judge_model` column. `--judge-effort`
|
|
70
72
|
(minimal|low|medium|high) sets `reasoning_effort` and requires a
|
|
71
73
|
provider-qualified judge; the Anthropic judge does not take one.
|
|
72
74
|
|
|
73
|
-
##
|
|
75
|
+
## Sweep
|
|
74
76
|
|
|
75
77
|
```sh
|
|
76
78
|
skillcheck sweep # only scenarios without results
|
|
77
79
|
skillcheck sweep --all # rerun everything
|
|
78
80
|
```
|
|
79
81
|
|
|
80
|
-
|
|
81
|
-
order, sequentially.
|
|
82
|
-
discovered.
|
|
82
|
+
Walks `<root>/skills/*/evals/*` and `<root>/cli/*/skills/*/evals/*`, in sorted
|
|
83
|
+
order, sequentially. A scenario needs both `task.md` and `criteria.json` to be
|
|
84
|
+
discovered. Exit 2 if anything errored, 1 if anything failed, else 0.
|
|
83
85
|
|
|
84
|
-
`EVALS_CONCURRENCY` is passed to promptfoo as `-j` (default 4).
|
|
86
|
+
`EVALS_CONCURRENCY` is passed to promptfoo as `-j` (default 4). It parallelizes
|
|
85
87
|
within one scenario, not across them.
|
|
86
88
|
|
|
87
|
-
|
|
88
|
-
layer ([uinaf/
|
|
89
|
-
|
|
89
|
+
One known failure mode: judge calls through a gateway can drop at the transport
|
|
90
|
+
layer ([uinaf/zebroid-infra#44](https://github.com/uinaf/zebroid-infra/issues/44)).
|
|
91
|
+
That surfaces as an ERROR with no usable result, not as a graded FAIL, and the
|
|
90
92
|
mitigation is a rerun. `sweep` without `--all` resumes, so a rerun only picks up
|
|
91
93
|
what is missing.
|
|
92
94
|
|
|
93
|
-
##
|
|
95
|
+
## Summarize
|
|
94
96
|
|
|
95
97
|
```sh
|
|
96
98
|
skillcheck summarize [--allow-mixed]
|
|
97
99
|
```
|
|
98
100
|
|
|
99
|
-
|
|
101
|
+
Reduces `<root>/.skillcheck/results/*.json` into
|
|
100
102
|
`<root>/.skillcheck/scorecards/<UTC-date>.json`: one entry per scenario with
|
|
101
103
|
skill, scenario, harness, tree sha, score, pass, both models, latency, tokens.
|
|
102
104
|
|
|
103
|
-
|
|
105
|
+
If a scorecard for today already exists, the two are merged on
|
|
104
106
|
`(skill, scenario, harness)`: entries from this run win, entries it did not
|
|
105
|
-
touch survive, and the merge is reported on stdout.
|
|
107
|
+
touch survive, and the merge is reported on stdout. Summarizing after rerunning
|
|
106
108
|
six of twenty-nine scenarios therefore leaves twenty-nine rows in the file, not
|
|
107
|
-
six.
|
|
109
|
+
six. A same-date file that cannot be parsed stops the write instead of being
|
|
108
110
|
overwritten.
|
|
109
111
|
|
|
110
|
-
|
|
111
|
-
failing the reduction.
|
|
112
|
+
Files that are not promptfoo results and ungraded transport errors are skipped
|
|
113
|
+
with a warning rather than failing the reduction. Graded assertion failures
|
|
114
|
+
remain scored results. If a skipped file matches an existing scorecard row,
|
|
115
|
+
summary generation fails and leaves the scorecard unchanged, so an errored rerun
|
|
116
|
+
cannot carry forward its old score. This also applies with `--allow-mixed`.
|
|
117
|
+
Runs keep a `<name>.json.attempt` marker until a graded result and its provenance
|
|
118
|
+
are written. An outstanding marker makes `summarize` skip that identity even
|
|
119
|
+
when the child produced no result file or left partial output. The marker does
|
|
120
|
+
not count as a result for the sweep's existence check, so no-output failures
|
|
121
|
+
remain eligible for retry.
|
|
112
122
|
|
|
113
|
-
##
|
|
123
|
+
## Provenance
|
|
114
124
|
|
|
115
|
-
|
|
125
|
+
Each successful run writes a `<name>.meta.json` sidecar next to its result:
|
|
116
126
|
|
|
117
127
|
```json
|
|
118
128
|
{
|
|
@@ -124,27 +134,28 @@ each successful run writes a `<name>.meta.json` sidecar next to its result:
|
|
|
124
134
|
```
|
|
125
135
|
|
|
126
136
|
`summarize` reads those sidecars and refuses to mix skills-tree revisions in one
|
|
127
|
-
scorecard
|
|
128
|
-
|
|
137
|
+
scorecard, including retained rows from partial reruns, unless `--allow-mixed`.
|
|
138
|
+
Rejection leaves the existing scorecard unchanged. With the override, the top-level `skills_tree_sha`
|
|
139
|
+
becomes `mixed` and per-entry shas remain. A result with no sidecar reduces as
|
|
129
140
|
`unattested`.
|
|
130
141
|
|
|
131
|
-
##
|
|
142
|
+
## State
|
|
132
143
|
|
|
133
144
|
`<root>/.skillcheck/` holds `scratch/` and `results/`, both disposable and safe
|
|
134
|
-
to gitignore, and `scorecards/`, which is meant to be committed.
|
|
145
|
+
to gitignore, and `scorecards/`, which is meant to be committed. Nothing is ever
|
|
135
146
|
written inside the installed package.
|
|
136
147
|
|
|
137
|
-
##
|
|
148
|
+
## Auth
|
|
138
149
|
|
|
139
|
-
|
|
|
150
|
+
| Variable | Effect |
|
|
140
151
|
| --------------------------------------------- | ----------------------------------------------------------------- |
|
|
141
|
-
| `ANTHROPIC_BASE_URL` + `ANTHROPIC_AUTH_TOKEN` | claude agent and judge go through a gateway
|
|
142
|
-
|
|
|
143
|
-
| `ANTHROPIC_API_KEY` |
|
|
144
|
-
| `CODEX_HOME` (default `~/.codex`) |
|
|
145
|
-
| `OPENAI_API_KEY` |
|
|
146
|
-
| `CURSOR_API_KEY` |
|
|
147
|
-
| `OPENAI_API_KEY` + `OPENAI_BASE_URL` |
|
|
148
|
-
|
|
149
|
-
|
|
152
|
+
| `ANTHROPIC_BASE_URL` + `ANTHROPIC_AUTH_TOKEN` | The claude agent and judge go through a gateway |
|
|
153
|
+
| None of the above | Falls back to the local Claude Code session |
|
|
154
|
+
| `ANTHROPIC_API_KEY` | Judge grades over `anthropic:messages:<model>` instead of the SDK |
|
|
155
|
+
| `CODEX_HOME` (default `~/.codex`) | Where the codex harness finds the local `codex` CLI login |
|
|
156
|
+
| `OPENAI_API_KEY` | Agent auth for codex when there is no local login |
|
|
157
|
+
| `CURSOR_API_KEY` | Agent auth for cursor; a logged-in `cursor-agent` also works |
|
|
158
|
+
| `OPENAI_API_KEY` + `OPENAI_BASE_URL` | A provider-qualified `--judge openai:…`, optionally via a gateway |
|
|
159
|
+
|
|
160
|
+
A bare `--judge` model stays on the Anthropic selection regardless of the
|
|
150
161
|
agent harness; a provider-qualified `--judge` uses that provider's env instead.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@uinaf/skillcheck",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.5.1",
|
|
4
4
|
"description": "Lint and eval harness for agent skills",
|
|
5
5
|
"homepage": "https://github.com/uinaf/skillcheck#readme",
|
|
6
6
|
"bugs": {
|
|
@@ -27,22 +27,37 @@
|
|
|
27
27
|
"registry": "https://registry.npmjs.org/"
|
|
28
28
|
},
|
|
29
29
|
"scripts": {
|
|
30
|
-
"verify": "vp
|
|
30
|
+
"verify": "vp run ready",
|
|
31
|
+
"verify:full": "vp run --no-cache ready",
|
|
31
32
|
"prepare": "vp config --no-agent",
|
|
32
|
-
"prepublishOnly": "pnpm run verify"
|
|
33
|
+
"prepublishOnly": "pnpm run verify:full"
|
|
33
34
|
},
|
|
34
|
-
"
|
|
35
|
+
"devDependencies": {
|
|
35
36
|
"@anthropic-ai/claude-agent-sdk": "^0.3.233",
|
|
36
37
|
"@openai/codex-sdk": "^0.147.0",
|
|
37
|
-
"promptfoo": "^0.122.0"
|
|
38
|
-
},
|
|
39
|
-
"devDependencies": {
|
|
40
38
|
"@types/node": "^26.2.0",
|
|
39
|
+
"promptfoo": "^0.122.0",
|
|
41
40
|
"vite": "catalog:",
|
|
42
41
|
"vite-plus": "catalog:"
|
|
43
42
|
},
|
|
43
|
+
"peerDependencies": {
|
|
44
|
+
"@anthropic-ai/claude-agent-sdk": "^0.3.233",
|
|
45
|
+
"@openai/codex-sdk": "^0.147.0",
|
|
46
|
+
"promptfoo": "^0.122.0"
|
|
47
|
+
},
|
|
48
|
+
"peerDependenciesMeta": {
|
|
49
|
+
"@anthropic-ai/claude-agent-sdk": {
|
|
50
|
+
"optional": true
|
|
51
|
+
},
|
|
52
|
+
"@openai/codex-sdk": {
|
|
53
|
+
"optional": true
|
|
54
|
+
},
|
|
55
|
+
"promptfoo": {
|
|
56
|
+
"optional": true
|
|
57
|
+
}
|
|
58
|
+
},
|
|
44
59
|
"engines": {
|
|
45
60
|
"node": ">=24"
|
|
46
61
|
},
|
|
47
|
-
"packageManager": "pnpm@
|
|
62
|
+
"packageManager": "pnpm@12.0.0"
|
|
48
63
|
}
|