@uinaf/skillcheck 0.1.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 undefined is not a function LLC
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md ADDED
@@ -0,0 +1,63 @@
1
+ ![skillcheck — lint and eval harness for agent skills.](https://uinaf.dev/og/banner/skillcheck.png)
2
+
3
+ # @uinaf/skillcheck
4
+
5
+ lint and eval harness for agent skills. one CLI with two halves: a keyless
6
+ structural lint any repo can run in CI, and a promptfoo-driven eval loop that
7
+ grades what a skill actually makes an agent do.
8
+
9
+ built for [uinaf](https://uinaf.dev) skill repos. nothing in it is
10
+ uinaf-specific. it ships no opinion about what a skill should say, only about
11
+ where skills sit and how a scenario is scored.
12
+
13
+ ## install
14
+
15
+ ```sh
16
+ pnpm add -D @uinaf/skillcheck
17
+ ```
18
+
19
+ node 24 or newer. the package ships compiled ESM and runs no install scripts,
20
+ so a runner using `--ignore-scripts` is fine. consumers still pinned to the
21
+ pre-npm git tags are covered in [adoption](docs/adoption.md).
22
+
23
+ ## use
24
+
25
+ ```sh
26
+ skillcheck lint # structural lint, no credentials
27
+ skillcheck run skills/wat/evals/basic # one scenario, graded end to end
28
+ skillcheck sweep && skillcheck summarize # every scenario, then a scorecard
29
+ ```
30
+
31
+ `lint` needs nothing. `run` and `sweep` need model auth, which is why sweeps
32
+ stay operator-run and consumer repos never hold credentials. every command,
33
+ flag, and auth variable is in [usage](docs/usage.md).
34
+
35
+ ## layout contract
36
+
37
+ frozen, not configurable. every command reads one root: `--root <dir>`, or the
38
+ current directory.
39
+
40
+ ```text
41
+ <root>/skills/<skill>/SKILL.md linted
42
+ <root>/skills/<skill>/evals/<scenario>/task.md the problem + input files
43
+ <root>/skills/<skill>/evals/<scenario>/criteria.json the weighted checklist
44
+ <root>/.skillcheck/ results, scratch, scorecards
45
+ ```
46
+
47
+ `cli/*/skills/<skill>/` is scanned too, for repos that keep a skill next to the
48
+ CLI it documents.
49
+
50
+ ## docs
51
+
52
+ | doc | when |
53
+ | ------------------------------- | ------------------------------------- |
54
+ | [usage](docs/usage.md) | every subcommand, flag, and auth path |
55
+ | [scenarios](docs/scenarios.md) | writing an eval scenario |
56
+ | [adoption](docs/adoption.md) | wiring the lint into another repo |
57
+ | [releasing](docs/releasing.md) | the npm pipeline |
58
+ | [contributing](CONTRIBUTING.md) | local setup and the verify gate |
59
+ | [security](SECURITY.md) | reporting a vulnerability |
60
+
61
+ ## license
62
+
63
+ MIT · undefined is not a function LLC
package/dist/cli.js ADDED
@@ -0,0 +1,344 @@
1
+ #!/usr/bin/env node
2
+ import { lintSkills } from "./lint.js";
3
+ import { generateRun, runNameFor } from "./scenario.js";
4
+ import { execFileSync, spawnSync } from "node:child_process";
5
+ import fs from "node:fs";
6
+ import path from "node:path";
7
+ import { fileURLToPath } from "node:url";
8
+ //#region src/cli.ts
9
+ const here = path.dirname(fileURLToPath(import.meta.url));
10
+ const packageDir = path.resolve(here, "..");
11
+ const selfExt = path.extname(fileURLToPath(import.meta.url));
12
+ function toolVersion() {
13
+ return JSON.parse(fs.readFileSync(path.join(packageDir, "package.json"), "utf8")).version;
14
+ }
15
+ function parseMaxTurns(raw) {
16
+ const n = Number(raw);
17
+ if (!Number.isInteger(n) || n <= 0) throw new Error(`--max-turns must be a positive integer, got ${JSON.stringify(raw)}`);
18
+ return n;
19
+ }
20
+ function fail(msg) {
21
+ console.error(msg);
22
+ process.exit(1);
23
+ }
24
+ function parseArgs(argv) {
25
+ const positional = [];
26
+ const flags = /* @__PURE__ */ new Map();
27
+ const takesValue = /* @__PURE__ */ new Set([
28
+ "--root",
29
+ "--agent",
30
+ "--judge",
31
+ "--harness",
32
+ "--max-turns"
33
+ ]);
34
+ for (let i = 0; i < argv.length; i++) {
35
+ const a = argv[i];
36
+ if (!a.startsWith("--")) positional.push(a);
37
+ else if (takesValue.has(a)) {
38
+ const v = argv[++i];
39
+ if (v === void 0 || v.startsWith("--")) throw new Error(`${a} needs a value`);
40
+ flags.set(a, v);
41
+ } else if (a === "--all" || a === "--allow-mixed") flags.set(a, true);
42
+ else throw new Error(`unknown flag: ${a}`);
43
+ }
44
+ return {
45
+ positional,
46
+ flags
47
+ };
48
+ }
49
+ function resolveRoot(flags) {
50
+ return path.resolve(flags.get("--root") ?? process.cwd());
51
+ }
52
+ function stateDirs(root) {
53
+ const base = path.join(root, ".skillcheck");
54
+ return {
55
+ results: path.join(base, "results"),
56
+ scratch: path.join(base, "scratch"),
57
+ scorecards: path.join(base, "scorecards")
58
+ };
59
+ }
60
+ function runOptions(flags) {
61
+ const harness = flags.get("--harness") ?? "claude";
62
+ if (harness !== "claude" && harness !== "codex") fail(`--harness must be claude or codex, got ${harness}`);
63
+ return {
64
+ harness,
65
+ agentModel: flags.get("--agent"),
66
+ judgeModel: flags.get("--judge") ?? "claude-opus-5",
67
+ maxTurns: flags.has("--max-turns") ? parseMaxTurns(flags.get("--max-turns")) : void 0
68
+ };
69
+ }
70
+ function classifyResult(raw) {
71
+ const root = raw;
72
+ const res = root?.results?.results?.[0];
73
+ if (res === void 0) return { error: "promptfoo output carried no result" };
74
+ const message = typeof res.error === "string" ? res.error.trim() : "";
75
+ if (message !== "") return { error: message };
76
+ const stats = root?.results?.stats;
77
+ if (stats !== void 0 && (stats.errors ?? 0) > 0 && (stats.successes ?? 0) === 0 && (stats.failures ?? 0) === 0) return { error: "promptfoo reported an errored test with nothing graded" };
78
+ if (typeof res.score !== "number" || typeof res.success !== "boolean") return { error: "promptfoo result carried no usable score" };
79
+ return {
80
+ score: res.score,
81
+ pass: res.success
82
+ };
83
+ }
84
+ function gitHead(root) {
85
+ return execFileSync("git", ["rev-parse", "HEAD"], {
86
+ cwd: root,
87
+ encoding: "utf8"
88
+ }).trim();
89
+ }
90
+ function metaPath(resultPath) {
91
+ return resultPath.replace(/\.json$/, ".meta.json");
92
+ }
93
+ function runScenario(scenarioDir, opts, root) {
94
+ const dirs = stateDirs(root);
95
+ const { name, configPath } = generateRun(path.resolve(scenarioDir), opts, {
96
+ scratchDir: dirs.scratch,
97
+ transformPath: path.join(here, `transform${selfExt}`)
98
+ });
99
+ fs.mkdirSync(dirs.results, { recursive: true });
100
+ const resultPath = path.join(dirs.results, `${name}.json`);
101
+ fs.rmSync(resultPath, { force: true });
102
+ fs.rmSync(metaPath(resultPath), { force: true });
103
+ const sha = gitHead(root);
104
+ const rc = spawnSync("npx", [
105
+ "promptfoo",
106
+ "eval",
107
+ "--no-cache",
108
+ "--no-progress-bar",
109
+ "-j",
110
+ process.env.EVALS_CONCURRENCY ?? "4",
111
+ "-c",
112
+ configPath,
113
+ "-o",
114
+ resultPath
115
+ ], {
116
+ cwd: packageDir,
117
+ stdio: "inherit",
118
+ env: {
119
+ ...process.env,
120
+ PROMPTFOO_FAILED_TEST_EXIT_CODE: "0"
121
+ }
122
+ }).status ?? 1;
123
+ const outcome = {
124
+ name,
125
+ rc,
126
+ resultPath
127
+ };
128
+ if (rc !== 0) return outcome;
129
+ let verdict;
130
+ try {
131
+ verdict = classifyResult(JSON.parse(fs.readFileSync(resultPath, "utf8")));
132
+ } catch {
133
+ verdict = { error: "promptfoo produced no parseable result file" };
134
+ }
135
+ outcome.score = verdict.score;
136
+ outcome.pass = verdict.pass;
137
+ outcome.error = verdict.error;
138
+ if (verdict.score !== void 0) fs.writeFileSync(metaPath(resultPath), JSON.stringify({
139
+ skills_tree_sha: sha,
140
+ harness: opts.harness,
141
+ ran_at: (/* @__PURE__ */ new Date()).toISOString(),
142
+ tool_version: toolVersion()
143
+ }, null, 2) + "\n");
144
+ return outcome;
145
+ }
146
+ function discoverScenarios(root) {
147
+ const roots = [path.join(root, "skills")];
148
+ const cliDir = path.join(root, "cli");
149
+ if (fs.existsSync(cliDir)) {
150
+ for (const e of fs.readdirSync(cliDir, { withFileTypes: true })) if (e.isDirectory()) roots.push(path.join(cliDir, e.name, "skills"));
151
+ }
152
+ const found = [];
153
+ for (const dir of roots) {
154
+ if (!fs.existsSync(dir)) continue;
155
+ for (const skill of fs.readdirSync(dir, { withFileTypes: true })) {
156
+ const evalsDir = path.join(dir, skill.name, "evals");
157
+ if (!skill.isDirectory() || !fs.existsSync(evalsDir)) continue;
158
+ for (const sc of fs.readdirSync(evalsDir, { withFileTypes: true })) {
159
+ const scenarioDir = path.join(evalsDir, sc.name);
160
+ if (sc.isDirectory() && fs.existsSync(path.join(scenarioDir, "task.md")) && fs.existsSync(path.join(scenarioDir, "criteria.json"))) found.push(scenarioDir);
161
+ }
162
+ }
163
+ }
164
+ return found.sort();
165
+ }
166
+ function cmdRun(argv) {
167
+ const { positional, flags } = parseArgs(argv);
168
+ if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--harness claude|codex]");
169
+ const o = runScenario(positional[0], runOptions(flags), resolveRoot(flags));
170
+ if (o.score === void 0) {
171
+ console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
172
+ process.exit(2);
173
+ }
174
+ console.log(`${o.pass ? "PASS" : "FAIL"} ${o.name} score=${o.score.toFixed(4)} (results: ${o.resultPath})`);
175
+ process.exit(o.pass ? 0 : 1);
176
+ }
177
+ function cmdSweep(argv) {
178
+ const { positional, flags } = parseArgs(argv);
179
+ if (positional.length > 0) fail("usage: skillcheck sweep [--root DIR] [--all]");
180
+ const root = resolveRoot(flags);
181
+ const opts = runOptions(flags);
182
+ const all = flags.get("--all") === true;
183
+ const resultsDir = stateDirs(root).results;
184
+ let passed = 0, failed = 0, errored = 0, skipped = 0;
185
+ for (const dir of discoverScenarios(root)) {
186
+ const name = runNameFor(dir, opts.harness);
187
+ const resultPath = path.join(resultsDir, `${name}.json`);
188
+ if (!all && fs.existsSync(resultPath)) {
189
+ skipped++;
190
+ console.log(`SKIP ${name} (results exist; use --all to rerun)`);
191
+ continue;
192
+ }
193
+ const o = runScenario(dir, opts, root);
194
+ if (o.score === void 0) {
195
+ errored++;
196
+ console.log(`ERROR ${o.name} ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
197
+ } else if (o.pass) {
198
+ passed++;
199
+ console.log(`PASS ${o.name} score=${o.score.toFixed(4)}`);
200
+ } else {
201
+ failed++;
202
+ console.log(`FAIL ${o.name} score=${o.score.toFixed(4)}`);
203
+ }
204
+ }
205
+ console.log(`\nsweep: ${passed} passed, ${failed} failed, ${errored} errored, ${skipped} skipped`);
206
+ process.exit(errored > 0 ? 2 : failed > 0 ? 1 : 0);
207
+ }
208
+ function reduceResults(dir, allowMixed) {
209
+ const entries = [];
210
+ const skipped = [];
211
+ const shas = /* @__PURE__ */ new Set();
212
+ for (const f of fs.readdirSync(dir).filter((f) => f.endsWith(".json") && !f.endsWith(".meta.json")).sort()) {
213
+ let raw;
214
+ try {
215
+ raw = JSON.parse(fs.readFileSync(path.join(dir, f), "utf8"));
216
+ } catch {
217
+ raw = void 0;
218
+ }
219
+ const res = raw?.results?.results?.[0];
220
+ if (typeof res?.score !== "number" || typeof res?.success !== "boolean") {
221
+ console.error(`skipping ${f}: not a promptfoo result`);
222
+ skipped.push(f);
223
+ continue;
224
+ }
225
+ const provider = raw.config?.providers?.[0];
226
+ const judge = raw.config?.defaultTest?.options?.provider;
227
+ const base = f.replace(/\.json$/, "");
228
+ const harness = base.endsWith("--codex") ? "codex" : "claude";
229
+ const [skill, ...rest] = base.replace(/--codex$/, "").split("--");
230
+ let sha = "unattested";
231
+ try {
232
+ sha = JSON.parse(fs.readFileSync(path.join(dir, `${base}.meta.json`), "utf8")).skills_tree_sha ?? "unattested";
233
+ } catch {}
234
+ shas.add(sha);
235
+ entries.push({
236
+ skill,
237
+ scenario: rest.join("--"),
238
+ harness,
239
+ skills_tree_sha: sha,
240
+ score: res.score,
241
+ pass: res.success,
242
+ agent_model: provider?.config?.model ?? "codex-default",
243
+ judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? "unknown",
244
+ latency_ms: res.latencyMs,
245
+ tokens: (res.tokenUsage?.total ?? 0) + (res.tokenUsage?.assertions?.total ?? 0)
246
+ });
247
+ }
248
+ if (shas.size > 1 && !allowMixed) throw new Error(`results span multiple skills-tree revisions (${[...shas].join(", ")}); rerun stale ones or pass --allow-mixed`);
249
+ return {
250
+ treeSha: shas.size === 1 ? [...shas][0] : shas.size === 0 ? "none" : "mixed",
251
+ entries,
252
+ skipped
253
+ };
254
+ }
255
+ function entryKey(e) {
256
+ return [
257
+ e.skill,
258
+ e.scenario,
259
+ e.harness
260
+ ].join("\0");
261
+ }
262
+ function mergeScorecard(existing, fresh) {
263
+ const byKey = /* @__PURE__ */ new Map();
264
+ for (const e of existing) byKey.set(entryKey(e), e);
265
+ let carried = byKey.size;
266
+ for (const e of fresh) {
267
+ if (byKey.has(entryKey(e))) carried--;
268
+ byKey.set(entryKey(e), e);
269
+ }
270
+ return {
271
+ entries: [...byKey.values()].sort((a, b) => entryKey(a) < entryKey(b) ? -1 : entryKey(a) > entryKey(b) ? 1 : 0),
272
+ carried
273
+ };
274
+ }
275
+ function treeShaOf(entries) {
276
+ const shas = new Set(entries.map((e) => e.skills_tree_sha));
277
+ return shas.size === 1 ? [...shas][0] : shas.size === 0 ? "none" : "mixed";
278
+ }
279
+ function readExistingScorecard(out) {
280
+ if (!fs.existsSync(out)) return [];
281
+ let prev;
282
+ try {
283
+ prev = JSON.parse(fs.readFileSync(out, "utf8"));
284
+ } catch {
285
+ throw new Error(`existing scorecard ${out} is not valid JSON; refusing to overwrite it`);
286
+ }
287
+ const scenarios = prev?.scenarios;
288
+ if (!Array.isArray(scenarios)) throw new Error(`existing scorecard ${out} has no scenarios array; refusing to overwrite it`);
289
+ return scenarios;
290
+ }
291
+ function cmdSummarize(argv) {
292
+ const { positional, flags } = parseArgs(argv);
293
+ if (positional.length > 0) fail("usage: skillcheck summarize [--root DIR] [--allow-mixed]");
294
+ const dirs = stateDirs(resolveRoot(flags));
295
+ if (!fs.existsSync(dirs.results)) fail(`no results directory at ${dirs.results} — run some evals first`);
296
+ const { entries, skipped } = reduceResults(dirs.results, flags.get("--allow-mixed") === true);
297
+ fs.mkdirSync(dirs.scorecards, { recursive: true });
298
+ const out = path.join(dirs.scorecards, `${(/* @__PURE__ */ new Date()).toISOString().slice(0, 10)}.json`);
299
+ const existing = readExistingScorecard(out);
300
+ const merged = mergeScorecard(existing, entries);
301
+ const scorecard = {
302
+ ran_at: (/* @__PURE__ */ new Date()).toISOString(),
303
+ skills_tree_sha: treeShaOf(merged.entries),
304
+ scenarios: merged.entries
305
+ };
306
+ fs.writeFileSync(out, JSON.stringify(scorecard, null, 2) + "\n");
307
+ console.log(`${out}: ${merged.entries.length} scenario(s), ${merged.entries.filter((e) => e.pass).length} passing, ${skipped.length} skipped file(s)`);
308
+ if (existing.length > 0) console.log(`merged into today's scorecard: ${entries.length} from this run, ${merged.carried} carried over`);
309
+ }
310
+ function cmdLint(argv) {
311
+ const { positional, flags } = parseArgs(argv);
312
+ if (positional.length > 1) fail("usage: skillcheck lint [<root>] [--root DIR]");
313
+ const root = positional.length === 1 ? path.resolve(positional[0]) : resolveRoot(flags);
314
+ const { errors, count } = lintSkills(root);
315
+ if (errors.length > 0) {
316
+ for (const e of errors) console.error(e);
317
+ console.error(`skill lint: ${errors.length} error(s) across ${count} package(s)`);
318
+ process.exit(1);
319
+ }
320
+ console.log(`skill lint: ${count} package(s) clean`);
321
+ }
322
+ function isMainModule() {
323
+ const entry = process.argv[1];
324
+ if (entry === void 0) return false;
325
+ try {
326
+ return fs.realpathSync(entry) === fs.realpathSync(fileURLToPath(import.meta.url));
327
+ } catch {
328
+ return false;
329
+ }
330
+ }
331
+ if (isMainModule()) {
332
+ const [cmd, ...rest] = process.argv.slice(2);
333
+ try {
334
+ if (cmd === "run") cmdRun(rest);
335
+ else if (cmd === "sweep") cmdSweep(rest);
336
+ else if (cmd === "summarize") cmdSummarize(rest);
337
+ else if (cmd === "lint") cmdLint(rest);
338
+ else fail("usage: skillcheck <lint|run|sweep|summarize> ...");
339
+ } catch (err) {
340
+ fail(err instanceof Error ? err.message : String(err));
341
+ }
342
+ }
343
+ //#endregion
344
+ export { classifyResult, mergeScorecard, parseArgs, parseMaxTurns, reduceResults, resolveRoot, stateDirs, toolVersion, treeShaOf };
package/dist/lint.js ADDED
@@ -0,0 +1,79 @@
1
+ import fs from "node:fs";
2
+ import path from "node:path";
3
+ //#region src/lint.ts
4
+ const ALLOWED_KEYS = /* @__PURE__ */ new Set([
5
+ "name",
6
+ "description",
7
+ "disable-model-invocation"
8
+ ]);
9
+ function lintSkill(dir, root, errors) {
10
+ const skillMd = path.join(dir, "SKILL.md");
11
+ const label = path.relative(root, skillMd);
12
+ if (!fs.existsSync(skillMd)) {
13
+ errors.push(`${path.relative(root, dir)}: missing SKILL.md`);
14
+ return;
15
+ }
16
+ const lines = fs.readFileSync(skillMd, "utf8").split("\n");
17
+ if (lines[0] !== "---") {
18
+ errors.push(`${label}: frontmatter must open with --- on line 1`);
19
+ return;
20
+ }
21
+ const close = lines.indexOf("---", 1);
22
+ if (close === -1) {
23
+ errors.push(`${label}: frontmatter never closes`);
24
+ return;
25
+ }
26
+ const fields = /* @__PURE__ */ new Map();
27
+ for (const line of lines.slice(1, close)) {
28
+ const m = line.match(/^([a-z-]+):\s*(.*)$/);
29
+ if (!m) {
30
+ errors.push(`${label}: unparseable frontmatter line: ${line}`);
31
+ continue;
32
+ }
33
+ const [, key, raw] = m;
34
+ if (!ALLOWED_KEYS.has(key)) {
35
+ errors.push(`${label}: unknown frontmatter key: ${key}`);
36
+ continue;
37
+ }
38
+ if (fields.has(key)) {
39
+ errors.push(`${label}: duplicate frontmatter key: ${key}`);
40
+ continue;
41
+ }
42
+ const value = raw.replace(/^"(.*)"$/s, "$1").replace(/^'(.*)'$/s, "$1").trim();
43
+ fields.set(key, {
44
+ raw: raw.trim(),
45
+ value
46
+ });
47
+ }
48
+ const name = fields.get("name")?.value ?? "";
49
+ if (name !== path.basename(dir)) errors.push(`${label}: frontmatter name ${JSON.stringify(name)} != directory ${path.basename(dir)}`);
50
+ if (!fields.get("description")?.value) errors.push(`${label}: description is required and must be non-empty`);
51
+ const dmi = fields.get("disable-model-invocation");
52
+ if (dmi !== void 0 && dmi.raw !== "true") errors.push(`${label}: disable-model-invocation must be the literal boolean true, got ${JSON.stringify(dmi.raw)}`);
53
+ const body = lines.slice(close + 1).join("\n").replace(/^```[\s\S]*?^```/gm, "").replace(/`[^`\n]*`/g, "");
54
+ for (const link of body.matchAll(/\]\(\s*(<[^>\n]*>|[^)\s]+)(?:\s+"[^"\n]*")?\s*\)/g)) {
55
+ const target = link[1].replace(/^<(.*)>$/, "$1");
56
+ if (/^[a-z][a-z+.-]*:/.test(target) || target.startsWith("#") || target === "") continue;
57
+ const resolved = path.join(dir, target.split("#")[0]);
58
+ if (!fs.existsSync(resolved)) errors.push(`${label}: link target does not exist: ${target}`);
59
+ }
60
+ }
61
+ function lintSkills(root) {
62
+ const roots = ["skills", ...fs.globSync("cli/*/skills", { cwd: root })].map((r) => path.join(root, r));
63
+ const errors = [];
64
+ let count = 0;
65
+ for (const dir of roots) {
66
+ if (!fs.existsSync(dir)) continue;
67
+ for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
68
+ if (!entry.isDirectory() || entry.name.startsWith(".")) continue;
69
+ lintSkill(path.join(dir, entry.name), root, errors);
70
+ count++;
71
+ }
72
+ }
73
+ return {
74
+ errors,
75
+ count
76
+ };
77
+ }
78
+ //#endregion
79
+ export { lintSkills };
@@ -0,0 +1,228 @@
1
+ import { createRequire } from "node:module";
2
+ import fs from "node:fs";
3
+ import path from "node:path";
4
+ import { createHash } from "node:crypto";
5
+ import os from "node:os";
6
+ //#region src/scenario.ts
7
+ function loadScenario(scenarioDir) {
8
+ const match = scenarioDir.match(/skills\/([^/]+)\/evals\/([^/]+)$/);
9
+ if (!match) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
10
+ const [, skill, scenario] = match;
11
+ const taskMd = fs.readFileSync(path.join(scenarioDir, "task.md"), "utf8");
12
+ const criteria = JSON.parse(fs.readFileSync(path.join(scenarioDir, "criteria.json"), "utf8"));
13
+ if (criteria.type !== "weighted_checklist" || !Array.isArray(criteria.checklist) || criteria.checklist.length === 0) throw new Error(`unsupported or empty criteria in ${scenarioDir}`);
14
+ for (const item of criteria.checklist) if (!(typeof item?.name === "string" && item.name.trim() !== "" && typeof item?.description === "string" && item.description.trim() !== "" && Number.isFinite(item?.max_score) && item.max_score > 0)) throw new Error(`invalid checklist item in ${scenarioDir}: ${JSON.stringify(item)}`);
15
+ const files = [];
16
+ let prompt = taskMd.replace(/^=+ FILE: (.+?) =+\n([\s\S]*?)\n=+ END FILE =+$/gm, (_, name, content) => {
17
+ files.push({
18
+ name: name.trim(),
19
+ content: content + "\n"
20
+ });
21
+ return `(Input file \`${name.trim()}\` is available in your working directory.)`;
22
+ });
23
+ const skillDir = path.resolve(scenarioDir, "../..");
24
+ if (isHiddenSkill(fs.readFileSync(path.join(skillDir, "SKILL.md"), "utf8"))) prompt = `Use the ${skill} skill for this task.\n\n${prompt}`;
25
+ return {
26
+ skill,
27
+ scenario,
28
+ name: `${skill}--${scenario}`,
29
+ skillDir,
30
+ prompt,
31
+ files,
32
+ criteria
33
+ };
34
+ }
35
+ function runNameFor(scenarioDir, harness) {
36
+ const m = path.resolve(scenarioDir).match(/skills\/([^/]+)\/evals\/([^/]+)$/);
37
+ if (!m) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
38
+ return harness === "codex" ? `${m[1]}--${m[2]}--codex` : `${m[1]}--${m[2]}`;
39
+ }
40
+ function frontmatterRange(text) {
41
+ const lines = text.split("\n");
42
+ if (lines[0] !== "---") return null;
43
+ const close = lines.indexOf("---", 1);
44
+ return close === -1 ? null : [1, close];
45
+ }
46
+ function isHiddenSkill(skillMd) {
47
+ const range = frontmatterRange(skillMd);
48
+ if (!range) return false;
49
+ return skillMd.split("\n").slice(range[0], range[1]).some((l) => l.startsWith("disable-model-invocation:"));
50
+ }
51
+ function stripHiddenFlag(skillMd) {
52
+ const range = frontmatterRange(skillMd);
53
+ if (!range) return skillMd;
54
+ return skillMd.split("\n").filter((l, i) => !(i >= range[0] && i < range[1] && l.startsWith("disable-model-invocation:"))).join("\n");
55
+ }
56
+ const RESERVED = /* @__PURE__ */ new Set([".claude", ".agents"]);
57
+ function materialize(s, runDir, harness) {
58
+ const workdir = path.join(runDir, "workdir");
59
+ fs.rmSync(runDir, {
60
+ recursive: true,
61
+ force: true
62
+ });
63
+ fs.mkdirSync(workdir, { recursive: true });
64
+ const seen = /* @__PURE__ */ new Set();
65
+ const planned = s.files.map((f) => {
66
+ const dest = path.resolve(workdir, f.name);
67
+ const rel = path.relative(workdir, dest);
68
+ if (rel === "" || rel.startsWith("..") || path.isAbsolute(rel)) throw new Error(`embedded file escapes workdir: ${f.name}`);
69
+ if (RESERVED.has(rel.split(path.sep)[0])) throw new Error(`embedded file targets reserved dir: ${f.name}`);
70
+ if (seen.has(dest)) throw new Error(`duplicate embedded file: ${f.name}`);
71
+ seen.add(dest);
72
+ return {
73
+ dest,
74
+ content: f.content
75
+ };
76
+ });
77
+ for (const { dest, content } of planned) {
78
+ fs.mkdirSync(path.dirname(dest), { recursive: true });
79
+ fs.writeFileSync(dest, content);
80
+ }
81
+ const roots = harness === "codex" ? [".claude", ".agents"] : [".claude"];
82
+ for (const root of roots) fs.cpSync(s.skillDir, path.join(workdir, root, "skills", s.skill), {
83
+ recursive: true,
84
+ filter: (src) => path.basename(src) !== "evals"
85
+ });
86
+ for (const root of roots) {
87
+ const skillMd = path.join(workdir, root, "skills", s.skill, "SKILL.md");
88
+ const text = fs.readFileSync(skillMd, "utf8");
89
+ const stripped = stripHiddenFlag(text);
90
+ if (stripped !== text) fs.writeFileSync(skillMd, stripped);
91
+ }
92
+ const manifest = {};
93
+ const walk = (dir) => {
94
+ for (const e of fs.readdirSync(dir, { withFileTypes: true })) {
95
+ const p = path.join(dir, e.name);
96
+ if (e.isDirectory()) {
97
+ if (e.name !== ".claude") walk(p);
98
+ } else manifest[path.relative(workdir, p)] = createHash("sha256").update(fs.readFileSync(p)).digest("hex");
99
+ }
100
+ };
101
+ walk(workdir);
102
+ const manifestPath = path.join(runDir, "manifest.json");
103
+ fs.writeFileSync(manifestPath, JSON.stringify(manifest, null, 2));
104
+ return {
105
+ workdir,
106
+ manifestPath
107
+ };
108
+ }
109
+ function agentProvider(opts, workdir, skill) {
110
+ if (opts.harness === "codex") return {
111
+ id: "openai:codex-sdk",
112
+ config: {
113
+ ...opts.agentModel ? { model: opts.agentModel } : {},
114
+ working_dir: workdir,
115
+ skip_git_repo_check: true,
116
+ enable_streaming: true,
117
+ sandbox_mode: "workspace-write",
118
+ cli_env: { CODEX_HOME: process.env.CODEX_HOME ?? path.join(os.homedir(), ".codex") }
119
+ }
120
+ };
121
+ return {
122
+ id: "anthropic:claude-agent-sdk",
123
+ config: {
124
+ model: opts.agentModel ?? "claude-opus-5",
125
+ apiKeyRequired: false,
126
+ working_dir: workdir,
127
+ setting_sources: ["project"],
128
+ skills: [skill],
129
+ permission_mode: "acceptEdits",
130
+ append_allowed_tools: [
131
+ "Read",
132
+ "Write",
133
+ "Edit",
134
+ "Glob",
135
+ "Grep"
136
+ ],
137
+ max_turns: opts.maxTurns ?? 50
138
+ }
139
+ };
140
+ }
141
+ function buildConfig(s, workdir, manifestPath, opts, transformPath) {
142
+ return {
143
+ description: `${s.skill}/${s.scenario}`,
144
+ prompts: ["{{task}}"],
145
+ providers: [agentProvider(opts, workdir, s.skill)],
146
+ defaultTest: { options: {
147
+ provider: process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
148
+ id: "anthropic:claude-agent-sdk",
149
+ config: {
150
+ model: opts.judgeModel,
151
+ apiKeyRequired: false,
152
+ max_turns: 3,
153
+ output_format: {
154
+ type: "json_schema",
155
+ schema: {
156
+ type: "object",
157
+ additionalProperties: false,
158
+ required: [
159
+ "reason",
160
+ "pass",
161
+ "score"
162
+ ],
163
+ properties: {
164
+ reason: { type: "string" },
165
+ pass: { type: "boolean" },
166
+ score: {
167
+ type: "number",
168
+ minimum: 0,
169
+ maximum: 1
170
+ }
171
+ }
172
+ }
173
+ }
174
+ }
175
+ },
176
+ transform: `file://${transformPath}`
177
+ } },
178
+ tests: [{
179
+ description: s.criteria.context,
180
+ vars: {
181
+ task: s.prompt,
182
+ workdir,
183
+ manifest: manifestPath
184
+ },
185
+ assert: [{
186
+ type: "assert-set",
187
+ threshold: .7,
188
+ assert: s.criteria.checklist.map((item) => ({
189
+ type: "llm-rubric",
190
+ value: `${item.name}: ${item.description}`,
191
+ weight: item.max_score
192
+ }))
193
+ }, {
194
+ type: "skill-used",
195
+ value: s.skill
196
+ }]
197
+ }]
198
+ };
199
+ }
200
+ const SDK_PACKAGES = ["@anthropic-ai/claude-agent-sdk", "@openai/codex-sdk"];
201
+ function sdkNodeModulesDir() {
202
+ const require = createRequire(import.meta.url);
203
+ for (const pkg of SDK_PACKAGES) for (const dir of require.resolve.paths(pkg) ?? []) if (fs.existsSync(path.join(dir, pkg, "package.json"))) return dir;
204
+ }
205
+ function generateRun(scenarioDir, opts, paths) {
206
+ const s = loadScenario(scenarioDir);
207
+ const name = runNameFor(scenarioDir, opts.harness);
208
+ const runDir = path.join(paths.scratchDir, name);
209
+ const { workdir, manifestPath } = materialize(s, runDir, opts.harness);
210
+ const sdkDir = sdkNodeModulesDir();
211
+ if (sdkDir !== void 0) {
212
+ const link = path.join(runDir, "node_modules");
213
+ fs.rmSync(link, {
214
+ recursive: true,
215
+ force: true
216
+ });
217
+ fs.symlinkSync(sdkDir, link, "dir");
218
+ }
219
+ const config = buildConfig(s, workdir, manifestPath, opts, paths.transformPath);
220
+ const configPath = path.join(runDir, "promptfooconfig.json");
221
+ fs.writeFileSync(configPath, JSON.stringify(config, null, 2));
222
+ return {
223
+ name,
224
+ configPath
225
+ };
226
+ }
227
+ //#endregion
228
+ export { buildConfig, generateRun, isHiddenSkill, loadScenario, materialize, runNameFor, sdkNodeModulesDir, stripHiddenFlag };
@@ -0,0 +1,67 @@
1
+ import fs from "node:fs";
2
+ import path from "node:path";
3
+ import { createHash } from "node:crypto";
4
+ //#region src/transform.ts
5
+ const PER_FILE_CAP = 4e3;
6
+ const TOTAL_CAP = 24e3;
7
+ function safeSlice(text, end) {
8
+ let cut = text.slice(0, end);
9
+ const last = cut.charCodeAt(cut.length - 1);
10
+ if (last >= 55296 && last <= 56319) cut = cut.slice(0, -1);
11
+ return cut;
12
+ }
13
+ function capped(rel, text) {
14
+ if (text.length <= PER_FILE_CAP) return `=== OUTPUT FILE: ${rel} ===\n${text}\n=== END OUTPUT FILE ===`;
15
+ return `=== OUTPUT FILE: ${rel} (truncated: showing ${PER_FILE_CAP} of ${text.length} chars) ===\n${safeSlice(text, PER_FILE_CAP)}\n=== END OUTPUT FILE ===`;
16
+ }
17
+ function transform(output, context) {
18
+ const workdir = context.vars.workdir;
19
+ const manifest = JSON.parse(fs.readFileSync(context.vars.manifest, "utf8"));
20
+ const sections = [];
21
+ const visited = /* @__PURE__ */ new Set();
22
+ const walk = (dir) => {
23
+ for (const e of fs.readdirSync(dir, { withFileTypes: true })) {
24
+ const p = path.join(dir, e.name);
25
+ if (e.isDirectory()) {
26
+ if (e.name !== ".claude") walk(p);
27
+ } else if (e.isFile()) {
28
+ const rel = path.relative(workdir, p);
29
+ visited.add(rel);
30
+ let body;
31
+ try {
32
+ body = fs.readFileSync(p);
33
+ } catch (err) {
34
+ const detail = err instanceof Error ? err.message : String(err);
35
+ sections.push({
36
+ rel,
37
+ text: `=== UNREADABLE FILE: ${rel} === (${detail})`
38
+ });
39
+ continue;
40
+ }
41
+ const hash = createHash("sha256").update(body).digest("hex");
42
+ if (manifest[rel] !== hash) sections.push({
43
+ rel,
44
+ text: capped(rel, body.toString("utf8"))
45
+ });
46
+ } else {
47
+ const nrel = path.relative(workdir, p);
48
+ sections.push({
49
+ rel: nrel,
50
+ text: `=== NON-REGULAR FILE: ${nrel} ===`
51
+ });
52
+ }
53
+ }
54
+ };
55
+ walk(workdir);
56
+ for (const rel of Object.keys(manifest)) if (!visited.has(rel)) sections.push({
57
+ rel,
58
+ text: `=== DELETED FILE: ${rel} === (input file removed by the agent)`
59
+ });
60
+ if (sections.length === 0) return output;
61
+ sections.sort((a, b) => a.rel < b.rel ? -1 : a.rel > b.rel ? 1 : 0);
62
+ let appended = sections.map((s) => s.text).join("\n\n");
63
+ if (appended.length > TOTAL_CAP) appended = `${safeSlice(appended, TOTAL_CAP)}\n\n=== TRUNCATED: output files exceeded ${TOTAL_CAP} chars total ===`;
64
+ return `${output}\n\n${appended}`;
65
+ }
66
+ //#endregion
67
+ export { transform as default };
@@ -0,0 +1,80 @@
1
+ # adopting it in a repo
2
+
3
+ two separate decisions: run the lint in CI, and run evals on a machine that has
4
+ model auth. only the first belongs in a consumer repo.
5
+
6
+ ## lint in CI
7
+
8
+ ```sh
9
+ pnpm add -D @uinaf/skillcheck
10
+ ```
11
+
12
+ runners that install with `--ignore-scripts` are fine: the package ships
13
+ compiled ESM and has no install, prepare, or postinstall script.
14
+
15
+ ```json
16
+ { "scripts": { "skills:lint": "skillcheck lint" } }
17
+ ```
18
+
19
+ ```yaml
20
+ jobs:
21
+ skills:
22
+ name: skills lint
23
+ runs-on: ubuntu-latest
24
+ steps:
25
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
26
+ with:
27
+ persist-credentials: false
28
+ - uses: actions/setup-node@249970729cb0ef3589644e2896645e5dc5ba9c38 # v6.5.0
29
+ with:
30
+ node-version: "24"
31
+ - run: pnpm install --frozen-lockfile
32
+ - run: pnpm run skills:lint
33
+ ```
34
+
35
+ the job needs no secrets and no network beyond the install. node 24 is the
36
+ floor.
37
+
38
+ run it through the script rather than a bare `npx skillcheck`: the script
39
+ resolves the version the repo pinned, and `npx` would resolve the latest one on
40
+ the registry.
41
+
42
+ ## evals
43
+
44
+ sweeps need model credentials, so they stay off consumer CI and run from an
45
+ operator machine or a job that already holds gateway auth:
46
+
47
+ ```sh
48
+ skillcheck sweep # resumes: only scenarios without results
49
+ skillcheck summarize # writes .skillcheck/scorecards/<UTC-date>.json
50
+ ```
51
+
52
+ commit `.skillcheck/scorecards/`. gitignore the rest:
53
+
54
+ ```gitignore
55
+ .skillcheck/results/
56
+ .skillcheck/scratch/
57
+ ```
58
+
59
+ a scorecard is only comparable against the tree it graded, which is why every
60
+ result carries the root repo's HEAD and `summarize` refuses to mix revisions
61
+ without `--allow-mixed`.
62
+
63
+ ## upgrading
64
+
65
+ bump the version in `package.json` and rerun the sweep. results carry the
66
+ `tool_version` that produced them, so a scorecard says which harness build it
67
+ came from as well as which skills tree.
68
+
69
+ ## the older install path
70
+
71
+ before the package was published, consumers installed it from a git tag:
72
+
73
+ ```sh
74
+ npm i -D github:uinaf/skillcheck#v0.1.3
75
+ ```
76
+
77
+ those tags are frozen and still work: they carry a committed `dist/`, and npm
78
+ 12 consumers needed `allow-git=root` in `.npmrc` to accept the spec at all. the
79
+ registry install needs none of that. tags from `v0.1.4` on are npm releases and
80
+ carry no `dist/`, so a git spec pointing at one will not run.
@@ -0,0 +1,80 @@
1
+ # releasing
2
+
3
+ ## pipeline
4
+
5
+ a push to `main` runs one workflow, `.github/workflows/release.yml`:
6
+
7
+ ```text
8
+ verify ──┐
9
+ ├──> release npm publish, OIDC + uinaf-releaser (release environment)
10
+ secrets ─┘
11
+ ```
12
+
13
+ `verify` and `secrets` are the shared gate, called from `verify.yml` and
14
+ `secrets.yml`, so one definition serves pull requests, the merge queue, and this
15
+ push. keep it that way: a second copy of the gate on a push-to-`main` workflow
16
+ races this one over the same commit.
17
+
18
+ the file name `release.yml` is load-bearing. see below.
19
+
20
+ ## npm
21
+
22
+ `@uinaf/skillcheck` publishes from `.github/workflows/release.yml` via npm
23
+ Trusted Publishing (OIDC) and the `uinaf-releaser` GitHub App. there is no npm
24
+ token in this repository, in its environments, or in the organization.
25
+
26
+ required on the `release` GitHub Environment:
27
+
28
+ | name | kind | purpose |
29
+ | ------------------------------- | ------ | ------------------------------------------- |
30
+ | `UINAF_RELEASE_APP_CLIENT_ID` | var | GitHub App client id for the releaser bot |
31
+ | `UINAF_RELEASE_APP_PRIVATE_KEY` | secret | GitHub App private key for the releaser bot |
32
+
33
+ the trusted publisher on npmjs.com is registered by **file path**, so
34
+ `.github/workflows/release.yml` cannot be renamed or moved without editing that
35
+ registration first. a rename fails the publish with an identity mismatch, and
36
+ nothing earlier in the run reports it. the `release` environment name is bound
37
+ the same way.
38
+
39
+ deleting the `release` environment deletes both rows above with it, and there is
40
+ no repo-level fallback: `create-github-app-token` then runs with empty inputs
41
+ and the job fails at that step. the private key cannot be read back from
42
+ GitHub; recreating it means generating a new one in the App settings.
43
+
44
+ ## version history
45
+
46
+ semantic-release owns the version and the tag. `tagFormat` is `v${version}` and
47
+ history continues from `v0.1.3`; `v0.1.0`–`v0.1.3` are the legacy git-install
48
+ tags and are never deleted or moved.
49
+
50
+ during preparation, `@semantic-release/npm` stages the released `package.json`
51
+ version and `@jno21/semantic-release-github-commit` commits it to `main` through
52
+ GitHub's API as the authenticated App. GitHub signs that commit, and the release
53
+ tag points to it. the `[skip ci]` marker on that commit is what stops a release
54
+ from releasing itself.
55
+
56
+ check what the next version would be without publishing anything:
57
+
58
+ ```sh
59
+ pnpm dlx semantic-release --dry-run --no-ci
60
+ ```
61
+
62
+ ## the artifact
63
+
64
+ `dist/` is generated and untracked. `prepublishOnly` runs `pnpm run verify`,
65
+ which builds it, so the tarball is always packed from a tree that just passed
66
+ the gate. `files` is `dist`, `docs`, `README.md`, `LICENSE`.
67
+
68
+ manual publish is emergency recovery only:
69
+
70
+ ```sh
71
+ pnpm run verify
72
+ npm publish --access public
73
+ ```
74
+
75
+ ## the old install path
76
+
77
+ before `@uinaf/skillcheck` existed, consumers installed
78
+ `github:uinaf/skillcheck#v0.1.3`. those tags still resolve and still carry a
79
+ committed `dist/`, so anything pinned to them keeps working untouched. new
80
+ consumers use npm.
@@ -0,0 +1,86 @@
1
+ # writing scenarios
2
+
3
+ a scenario is two files in a frozen location:
4
+
5
+ ```text
6
+ <root>/skills/<skill>/evals/<scenario>/task.md
7
+ <root>/skills/<skill>/evals/<scenario>/criteria.json
8
+ ```
9
+
10
+ the path is the identity: `<skill>--<scenario>` names the run, the result file,
11
+ and the scorecard entry. on the codex harness the name gains a `--codex` suffix,
12
+ so both harnesses can hold results side by side. a directory missing either file
13
+ is not discovered.
14
+
15
+ ## task.md
16
+
17
+ the prompt handed to the agent, verbatim, with one piece of syntax. input files
18
+ are embedded inline and materialized into the workdir before the run:
19
+
20
+ ```md
21
+ Fix the failing check in the config below.
22
+
23
+ ======= FILE: config.json =======
24
+ { "retries": -1 }
25
+ ======= END FILE =======
26
+ ```
27
+
28
+ each block is replaced in the prompt with a pointer ("Input file `config.json`
29
+ is available in your working directory.") and written to disk. destinations
30
+ must stay under the workdir, must not collide, and must not target `.claude/` or
31
+ `.agents/`, since a fixture that writes agent config would be configuring its
32
+ own examiner.
33
+
34
+ write the task the way a user would write it. do not name the skill, describe
35
+ its steps, or hint at the checklist: routing is part of what is being measured.
36
+
37
+ ## criteria.json
38
+
39
+ ```json
40
+ {
41
+ "type": "weighted_checklist",
42
+ "context": "one line describing what a good answer looks like",
43
+ "checklist": [
44
+ { "name": "short-handle", "description": "what the judge should look for", "max_score": 3 }
45
+ ]
46
+ }
47
+ ```
48
+
49
+ `type` must be `weighted_checklist` and the checklist must be non-empty. every
50
+ item needs a non-empty `name` and `description` and a positive `max_score`.
51
+
52
+ each item becomes one `llm-rubric` assertion weighted by `max_score`, inside an
53
+ assert-set with threshold 0.7. a separate `skill-used` assertion sits outside
54
+ that aggregate, so a run that produces good output without ever loading the
55
+ skill still fails. there is no test-level threshold: both must pass.
56
+
57
+ write descriptions a judge can check against the deliverable: an observable
58
+ property, not a feeling. weight the items that would make a reviewer reject the
59
+ work.
60
+
61
+ ## what the judge sees
62
+
63
+ the agent's final message, plus every file in the workdir that differs from the
64
+ pre-run manifest. unchanged inputs are omitted; deleted inputs, unreadable
65
+ files, and non-regular files are named rather than read.
66
+
67
+ sections are sorted by path, each file is capped at 4,000 characters and the
68
+ appended total at 24,000, with truncation stated inline. very large outputs make
69
+ rubric judges return nothing at all, which is why the caps exist. keep fixtures
70
+ small enough that the deliverable fits.
71
+
72
+ ## hidden skills
73
+
74
+ a skill with `disable-model-invocation: true` is explicit-invoke-only in
75
+ production, which the agent SDK cannot simulate. so the eval copy, never the shipped one,
76
+ has the flag stripped, and the task gains a leading
77
+ `Use the <skill> skill for this task.` the eval then measures
78
+ behavior-when-invoked rather than routing. the flag is only honored inside the
79
+ frontmatter block; body text mentioning the key does not count.
80
+
81
+ ## the workdir
82
+
83
+ per run, under `<root>/.skillcheck/scratch/<name>/`, rebuilt from scratch each
84
+ time. the skill under test is installed at `.claude/skills/<skill>/` (and also
85
+ `.agents/skills/<skill>/` on the codex harness) with its `evals/` directory
86
+ excluded, so criteria never leak into the agent's context.
package/docs/usage.md ADDED
@@ -0,0 +1,125 @@
1
+ # usage
2
+
3
+ every subcommand resolves one root, `--root <dir>` or the current directory.
4
+ `lint` also takes the root as a positional, because that is the shape CI reaches
5
+ for first.
6
+
7
+ ## lint
8
+
9
+ ```sh
10
+ skillcheck lint # lints the current repo
11
+ skillcheck lint ../other # lints another root
12
+ ```
13
+
14
+ checks each `<root>/skills/<skill>/`:
15
+
16
+ - frontmatter opens with `---` on line 1 and closes
17
+ - keys are `name`, `description`, `disable-model-invocation` and nothing else,
18
+ each at most once
19
+ - `name` equals the directory name; `description` is non-empty
20
+ - `disable-model-invocation`, when present, is the bare YAML boolean `true`.
21
+ a quoted `"true"` is an error
22
+ - relative links in the body resolve on disk
23
+
24
+ code spans and fenced blocks are stripped before links are checked, so example
25
+ links never fail. external schemes and `#anchors` pass. dot-directories under
26
+ `skills/` (`.claude-plugin`) are plugin metadata, not packages, and are skipped.
27
+
28
+ findings print one per line, relative to the linted root, then a count. exit 0
29
+ clean, 1 with findings.
30
+
31
+ ## run
32
+
33
+ ```sh
34
+ skillcheck run skills/<skill>/evals/<scenario>
35
+ skillcheck run <scenario-dir> --agent MODEL --judge MODEL --harness codex --max-turns 80
36
+ ```
37
+
38
+ materializes the scenario into `<root>/.skillcheck/scratch/<name>/workdir`,
39
+ installs the skill under test into that workdir, drives the agent, and grades
40
+ the files it wrote. exit 0 pass, 1 graded fail, 2 error (promptfoo produced no
41
+ usable result).
42
+
43
+ a test that errored was never graded, so it exits 2, prints the provider's
44
+ message, and writes no provenance sidecar. it is never reported as
45
+ `FAIL score=0.0000`; only a real judged verdict can fail a run.
46
+
47
+ defaults: `--harness claude`, agent `claude-opus-5`, judge `claude-opus-5`,
48
+ `--max-turns 50`. on the codex harness, omitting `--agent` leaves the model to
49
+ the Codex CLI's own default.
50
+
51
+ ## sweep
52
+
53
+ ```sh
54
+ skillcheck sweep # only scenarios without results
55
+ skillcheck sweep --all # rerun everything
56
+ ```
57
+
58
+ walks `<root>/skills/*/evals/*` and `<root>/cli/*/skills/*/evals/*`, in sorted
59
+ order, sequentially. a scenario needs both `task.md` and `criteria.json` to be
60
+ discovered. exit 2 if anything errored, 1 if anything failed, else 0.
61
+
62
+ `EVALS_CONCURRENCY` is passed to promptfoo as `-j` (default 4). it parallelizes
63
+ within one scenario, not across them.
64
+
65
+ one known failure mode: judge calls through a gateway can drop at the transport
66
+ layer ([uinaf/agent-platform#28](https://github.com/uinaf/agent-platform/issues/28)).
67
+ that surfaces as an ERROR with no usable result, not as a graded FAIL, and the
68
+ mitigation is a rerun. `sweep` without `--all` resumes, so a rerun only picks up
69
+ what is missing.
70
+
71
+ ## summarize
72
+
73
+ ```sh
74
+ skillcheck summarize [--allow-mixed]
75
+ ```
76
+
77
+ reduces `<root>/.skillcheck/results/*.json` into
78
+ `<root>/.skillcheck/scorecards/<UTC-date>.json`: one entry per scenario with
79
+ skill, scenario, harness, tree sha, score, pass, both models, latency, tokens.
80
+
81
+ if a scorecard for today already exists, the two are merged on
82
+ `(skill, scenario, harness)`: entries from this run win, entries it did not
83
+ touch survive, and the merge is reported on stdout. summarizing after rerunning
84
+ six of twenty-nine scenarios therefore leaves twenty-nine rows in the file, not
85
+ six. a same-date file that cannot be parsed stops the write instead of being
86
+ overwritten.
87
+
88
+ files that are not promptfoo results are skipped with a warning rather than
89
+ failing the reduction.
90
+
91
+ ## provenance
92
+
93
+ each successful run writes a `<name>.meta.json` sidecar next to its result:
94
+
95
+ ```json
96
+ {
97
+ "skills_tree_sha": "<root repo HEAD at run time>",
98
+ "harness": "claude",
99
+ "ran_at": "<ISO timestamp>",
100
+ "tool_version": "<skillcheck version>"
101
+ }
102
+ ```
103
+
104
+ `summarize` reads those sidecars and refuses to mix skills-tree revisions in one
105
+ scorecard unless `--allow-mixed`, in which case the top-level `skills_tree_sha`
106
+ becomes `mixed` and per-entry shas remain. a result with no sidecar reduces as
107
+ `unattested`.
108
+
109
+ ## state
110
+
111
+ `<root>/.skillcheck/` holds `scratch/` and `results/`, both disposable and safe
112
+ to gitignore, and `scorecards/`, which is meant to be committed. nothing is ever
113
+ written inside the installed package.
114
+
115
+ ## auth
116
+
117
+ | variable | effect |
118
+ | --------------------------------------------- | ----------------------------------------------------------------- |
119
+ | `ANTHROPIC_BASE_URL` + `ANTHROPIC_AUTH_TOKEN` | claude agent and judge go through a gateway |
120
+ | none of the above | falls back to the local Claude Code session |
121
+ | `ANTHROPIC_API_KEY` | judge grades over `anthropic:messages:<model>` instead of the SDK |
122
+ | `CODEX_HOME` (default `~/.codex`) | where the codex harness finds the local `codex` CLI login |
123
+ | `OPENAI_API_KEY` | codex agent auth when there is no local login |
124
+
125
+ the judge stays on the Anthropic selection regardless of the agent harness.
package/package.json ADDED
@@ -0,0 +1,48 @@
1
+ {
2
+ "name": "@uinaf/skillcheck",
3
+ "version": "0.1.3",
4
+ "description": "Lint and eval harness for agent skills",
5
+ "homepage": "https://github.com/uinaf/skillcheck#readme",
6
+ "bugs": {
7
+ "url": "https://github.com/uinaf/skillcheck/issues"
8
+ },
9
+ "license": "MIT",
10
+ "author": "undefined is not a function LLC",
11
+ "repository": {
12
+ "type": "git",
13
+ "url": "git+https://github.com/uinaf/skillcheck.git"
14
+ },
15
+ "bin": {
16
+ "skillcheck": "dist/cli.js"
17
+ },
18
+ "files": [
19
+ "dist",
20
+ "docs",
21
+ "README.md",
22
+ "LICENSE"
23
+ ],
24
+ "type": "module",
25
+ "publishConfig": {
26
+ "access": "public",
27
+ "registry": "https://registry.npmjs.org/"
28
+ },
29
+ "scripts": {
30
+ "verify": "vp check && vp pack && vp test run",
31
+ "prepare": "vp config --no-agent",
32
+ "prepublishOnly": "pnpm run verify"
33
+ },
34
+ "dependencies": {
35
+ "@anthropic-ai/claude-agent-sdk": "^0.3.233",
36
+ "@openai/codex-sdk": "^0.147.0",
37
+ "promptfoo": "^0.122.0"
38
+ },
39
+ "devDependencies": {
40
+ "@types/node": "^26.2.0",
41
+ "vite": "catalog:",
42
+ "vite-plus": "catalog:"
43
+ },
44
+ "engines": {
45
+ "node": ">=24"
46
+ },
47
+ "packageManager": "pnpm@11.22.0"
48
+ }