@uinaf/skillcheck 0.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +63 -0
- package/dist/cli.js +344 -0
- package/dist/lint.js +79 -0
- package/dist/scenario.js +228 -0
- package/dist/transform.js +67 -0
- package/docs/adoption.md +80 -0
- package/docs/releasing.md +80 -0
- package/docs/scenarios.md +86 -0
- package/docs/usage.md +125 -0
- package/package.json +48 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 undefined is not a function LLC
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+

|
|
2
|
+
|
|
3
|
+
# @uinaf/skillcheck
|
|
4
|
+
|
|
5
|
+
lint and eval harness for agent skills. one CLI with two halves: a keyless
|
|
6
|
+
structural lint any repo can run in CI, and a promptfoo-driven eval loop that
|
|
7
|
+
grades what a skill actually makes an agent do.
|
|
8
|
+
|
|
9
|
+
built for [uinaf](https://uinaf.dev) skill repos. nothing in it is
|
|
10
|
+
uinaf-specific. it ships no opinion about what a skill should say, only about
|
|
11
|
+
where skills sit and how a scenario is scored.
|
|
12
|
+
|
|
13
|
+
## install
|
|
14
|
+
|
|
15
|
+
```sh
|
|
16
|
+
pnpm add -D @uinaf/skillcheck
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
node 24 or newer. the package ships compiled ESM and runs no install scripts,
|
|
20
|
+
so a runner using `--ignore-scripts` is fine. consumers still pinned to the
|
|
21
|
+
pre-npm git tags are covered in [adoption](docs/adoption.md).
|
|
22
|
+
|
|
23
|
+
## use
|
|
24
|
+
|
|
25
|
+
```sh
|
|
26
|
+
skillcheck lint # structural lint, no credentials
|
|
27
|
+
skillcheck run skills/wat/evals/basic # one scenario, graded end to end
|
|
28
|
+
skillcheck sweep && skillcheck summarize # every scenario, then a scorecard
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
`lint` needs nothing. `run` and `sweep` need model auth, which is why sweeps
|
|
32
|
+
stay operator-run and consumer repos never hold credentials. every command,
|
|
33
|
+
flag, and auth variable is in [usage](docs/usage.md).
|
|
34
|
+
|
|
35
|
+
## layout contract
|
|
36
|
+
|
|
37
|
+
frozen, not configurable. every command reads one root: `--root <dir>`, or the
|
|
38
|
+
current directory.
|
|
39
|
+
|
|
40
|
+
```text
|
|
41
|
+
<root>/skills/<skill>/SKILL.md linted
|
|
42
|
+
<root>/skills/<skill>/evals/<scenario>/task.md the problem + input files
|
|
43
|
+
<root>/skills/<skill>/evals/<scenario>/criteria.json the weighted checklist
|
|
44
|
+
<root>/.skillcheck/ results, scratch, scorecards
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
`cli/*/skills/<skill>/` is scanned too, for repos that keep a skill next to the
|
|
48
|
+
CLI it documents.
|
|
49
|
+
|
|
50
|
+
## docs
|
|
51
|
+
|
|
52
|
+
| doc | when |
|
|
53
|
+
| ------------------------------- | ------------------------------------- |
|
|
54
|
+
| [usage](docs/usage.md) | every subcommand, flag, and auth path |
|
|
55
|
+
| [scenarios](docs/scenarios.md) | writing an eval scenario |
|
|
56
|
+
| [adoption](docs/adoption.md) | wiring the lint into another repo |
|
|
57
|
+
| [releasing](docs/releasing.md) | the npm pipeline |
|
|
58
|
+
| [contributing](CONTRIBUTING.md) | local setup and the verify gate |
|
|
59
|
+
| [security](SECURITY.md) | reporting a vulnerability |
|
|
60
|
+
|
|
61
|
+
## license
|
|
62
|
+
|
|
63
|
+
MIT · undefined is not a function LLC
|
package/dist/cli.js
ADDED
|
@@ -0,0 +1,344 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { lintSkills } from "./lint.js";
|
|
3
|
+
import { generateRun, runNameFor } from "./scenario.js";
|
|
4
|
+
import { execFileSync, spawnSync } from "node:child_process";
|
|
5
|
+
import fs from "node:fs";
|
|
6
|
+
import path from "node:path";
|
|
7
|
+
import { fileURLToPath } from "node:url";
|
|
8
|
+
//#region src/cli.ts
|
|
9
|
+
const here = path.dirname(fileURLToPath(import.meta.url));
|
|
10
|
+
const packageDir = path.resolve(here, "..");
|
|
11
|
+
const selfExt = path.extname(fileURLToPath(import.meta.url));
|
|
12
|
+
function toolVersion() {
|
|
13
|
+
return JSON.parse(fs.readFileSync(path.join(packageDir, "package.json"), "utf8")).version;
|
|
14
|
+
}
|
|
15
|
+
function parseMaxTurns(raw) {
|
|
16
|
+
const n = Number(raw);
|
|
17
|
+
if (!Number.isInteger(n) || n <= 0) throw new Error(`--max-turns must be a positive integer, got ${JSON.stringify(raw)}`);
|
|
18
|
+
return n;
|
|
19
|
+
}
|
|
20
|
+
function fail(msg) {
|
|
21
|
+
console.error(msg);
|
|
22
|
+
process.exit(1);
|
|
23
|
+
}
|
|
24
|
+
function parseArgs(argv) {
|
|
25
|
+
const positional = [];
|
|
26
|
+
const flags = /* @__PURE__ */ new Map();
|
|
27
|
+
const takesValue = /* @__PURE__ */ new Set([
|
|
28
|
+
"--root",
|
|
29
|
+
"--agent",
|
|
30
|
+
"--judge",
|
|
31
|
+
"--harness",
|
|
32
|
+
"--max-turns"
|
|
33
|
+
]);
|
|
34
|
+
for (let i = 0; i < argv.length; i++) {
|
|
35
|
+
const a = argv[i];
|
|
36
|
+
if (!a.startsWith("--")) positional.push(a);
|
|
37
|
+
else if (takesValue.has(a)) {
|
|
38
|
+
const v = argv[++i];
|
|
39
|
+
if (v === void 0 || v.startsWith("--")) throw new Error(`${a} needs a value`);
|
|
40
|
+
flags.set(a, v);
|
|
41
|
+
} else if (a === "--all" || a === "--allow-mixed") flags.set(a, true);
|
|
42
|
+
else throw new Error(`unknown flag: ${a}`);
|
|
43
|
+
}
|
|
44
|
+
return {
|
|
45
|
+
positional,
|
|
46
|
+
flags
|
|
47
|
+
};
|
|
48
|
+
}
|
|
49
|
+
function resolveRoot(flags) {
|
|
50
|
+
return path.resolve(flags.get("--root") ?? process.cwd());
|
|
51
|
+
}
|
|
52
|
+
function stateDirs(root) {
|
|
53
|
+
const base = path.join(root, ".skillcheck");
|
|
54
|
+
return {
|
|
55
|
+
results: path.join(base, "results"),
|
|
56
|
+
scratch: path.join(base, "scratch"),
|
|
57
|
+
scorecards: path.join(base, "scorecards")
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
function runOptions(flags) {
|
|
61
|
+
const harness = flags.get("--harness") ?? "claude";
|
|
62
|
+
if (harness !== "claude" && harness !== "codex") fail(`--harness must be claude or codex, got ${harness}`);
|
|
63
|
+
return {
|
|
64
|
+
harness,
|
|
65
|
+
agentModel: flags.get("--agent"),
|
|
66
|
+
judgeModel: flags.get("--judge") ?? "claude-opus-5",
|
|
67
|
+
maxTurns: flags.has("--max-turns") ? parseMaxTurns(flags.get("--max-turns")) : void 0
|
|
68
|
+
};
|
|
69
|
+
}
|
|
70
|
+
function classifyResult(raw) {
|
|
71
|
+
const root = raw;
|
|
72
|
+
const res = root?.results?.results?.[0];
|
|
73
|
+
if (res === void 0) return { error: "promptfoo output carried no result" };
|
|
74
|
+
const message = typeof res.error === "string" ? res.error.trim() : "";
|
|
75
|
+
if (message !== "") return { error: message };
|
|
76
|
+
const stats = root?.results?.stats;
|
|
77
|
+
if (stats !== void 0 && (stats.errors ?? 0) > 0 && (stats.successes ?? 0) === 0 && (stats.failures ?? 0) === 0) return { error: "promptfoo reported an errored test with nothing graded" };
|
|
78
|
+
if (typeof res.score !== "number" || typeof res.success !== "boolean") return { error: "promptfoo result carried no usable score" };
|
|
79
|
+
return {
|
|
80
|
+
score: res.score,
|
|
81
|
+
pass: res.success
|
|
82
|
+
};
|
|
83
|
+
}
|
|
84
|
+
function gitHead(root) {
|
|
85
|
+
return execFileSync("git", ["rev-parse", "HEAD"], {
|
|
86
|
+
cwd: root,
|
|
87
|
+
encoding: "utf8"
|
|
88
|
+
}).trim();
|
|
89
|
+
}
|
|
90
|
+
function metaPath(resultPath) {
|
|
91
|
+
return resultPath.replace(/\.json$/, ".meta.json");
|
|
92
|
+
}
|
|
93
|
+
function runScenario(scenarioDir, opts, root) {
|
|
94
|
+
const dirs = stateDirs(root);
|
|
95
|
+
const { name, configPath } = generateRun(path.resolve(scenarioDir), opts, {
|
|
96
|
+
scratchDir: dirs.scratch,
|
|
97
|
+
transformPath: path.join(here, `transform${selfExt}`)
|
|
98
|
+
});
|
|
99
|
+
fs.mkdirSync(dirs.results, { recursive: true });
|
|
100
|
+
const resultPath = path.join(dirs.results, `${name}.json`);
|
|
101
|
+
fs.rmSync(resultPath, { force: true });
|
|
102
|
+
fs.rmSync(metaPath(resultPath), { force: true });
|
|
103
|
+
const sha = gitHead(root);
|
|
104
|
+
const rc = spawnSync("npx", [
|
|
105
|
+
"promptfoo",
|
|
106
|
+
"eval",
|
|
107
|
+
"--no-cache",
|
|
108
|
+
"--no-progress-bar",
|
|
109
|
+
"-j",
|
|
110
|
+
process.env.EVALS_CONCURRENCY ?? "4",
|
|
111
|
+
"-c",
|
|
112
|
+
configPath,
|
|
113
|
+
"-o",
|
|
114
|
+
resultPath
|
|
115
|
+
], {
|
|
116
|
+
cwd: packageDir,
|
|
117
|
+
stdio: "inherit",
|
|
118
|
+
env: {
|
|
119
|
+
...process.env,
|
|
120
|
+
PROMPTFOO_FAILED_TEST_EXIT_CODE: "0"
|
|
121
|
+
}
|
|
122
|
+
}).status ?? 1;
|
|
123
|
+
const outcome = {
|
|
124
|
+
name,
|
|
125
|
+
rc,
|
|
126
|
+
resultPath
|
|
127
|
+
};
|
|
128
|
+
if (rc !== 0) return outcome;
|
|
129
|
+
let verdict;
|
|
130
|
+
try {
|
|
131
|
+
verdict = classifyResult(JSON.parse(fs.readFileSync(resultPath, "utf8")));
|
|
132
|
+
} catch {
|
|
133
|
+
verdict = { error: "promptfoo produced no parseable result file" };
|
|
134
|
+
}
|
|
135
|
+
outcome.score = verdict.score;
|
|
136
|
+
outcome.pass = verdict.pass;
|
|
137
|
+
outcome.error = verdict.error;
|
|
138
|
+
if (verdict.score !== void 0) fs.writeFileSync(metaPath(resultPath), JSON.stringify({
|
|
139
|
+
skills_tree_sha: sha,
|
|
140
|
+
harness: opts.harness,
|
|
141
|
+
ran_at: (/* @__PURE__ */ new Date()).toISOString(),
|
|
142
|
+
tool_version: toolVersion()
|
|
143
|
+
}, null, 2) + "\n");
|
|
144
|
+
return outcome;
|
|
145
|
+
}
|
|
146
|
+
function discoverScenarios(root) {
|
|
147
|
+
const roots = [path.join(root, "skills")];
|
|
148
|
+
const cliDir = path.join(root, "cli");
|
|
149
|
+
if (fs.existsSync(cliDir)) {
|
|
150
|
+
for (const e of fs.readdirSync(cliDir, { withFileTypes: true })) if (e.isDirectory()) roots.push(path.join(cliDir, e.name, "skills"));
|
|
151
|
+
}
|
|
152
|
+
const found = [];
|
|
153
|
+
for (const dir of roots) {
|
|
154
|
+
if (!fs.existsSync(dir)) continue;
|
|
155
|
+
for (const skill of fs.readdirSync(dir, { withFileTypes: true })) {
|
|
156
|
+
const evalsDir = path.join(dir, skill.name, "evals");
|
|
157
|
+
if (!skill.isDirectory() || !fs.existsSync(evalsDir)) continue;
|
|
158
|
+
for (const sc of fs.readdirSync(evalsDir, { withFileTypes: true })) {
|
|
159
|
+
const scenarioDir = path.join(evalsDir, sc.name);
|
|
160
|
+
if (sc.isDirectory() && fs.existsSync(path.join(scenarioDir, "task.md")) && fs.existsSync(path.join(scenarioDir, "criteria.json"))) found.push(scenarioDir);
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
return found.sort();
|
|
165
|
+
}
|
|
166
|
+
function cmdRun(argv) {
|
|
167
|
+
const { positional, flags } = parseArgs(argv);
|
|
168
|
+
if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--harness claude|codex]");
|
|
169
|
+
const o = runScenario(positional[0], runOptions(flags), resolveRoot(flags));
|
|
170
|
+
if (o.score === void 0) {
|
|
171
|
+
console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
|
|
172
|
+
process.exit(2);
|
|
173
|
+
}
|
|
174
|
+
console.log(`${o.pass ? "PASS" : "FAIL"} ${o.name} score=${o.score.toFixed(4)} (results: ${o.resultPath})`);
|
|
175
|
+
process.exit(o.pass ? 0 : 1);
|
|
176
|
+
}
|
|
177
|
+
function cmdSweep(argv) {
|
|
178
|
+
const { positional, flags } = parseArgs(argv);
|
|
179
|
+
if (positional.length > 0) fail("usage: skillcheck sweep [--root DIR] [--all]");
|
|
180
|
+
const root = resolveRoot(flags);
|
|
181
|
+
const opts = runOptions(flags);
|
|
182
|
+
const all = flags.get("--all") === true;
|
|
183
|
+
const resultsDir = stateDirs(root).results;
|
|
184
|
+
let passed = 0, failed = 0, errored = 0, skipped = 0;
|
|
185
|
+
for (const dir of discoverScenarios(root)) {
|
|
186
|
+
const name = runNameFor(dir, opts.harness);
|
|
187
|
+
const resultPath = path.join(resultsDir, `${name}.json`);
|
|
188
|
+
if (!all && fs.existsSync(resultPath)) {
|
|
189
|
+
skipped++;
|
|
190
|
+
console.log(`SKIP ${name} (results exist; use --all to rerun)`);
|
|
191
|
+
continue;
|
|
192
|
+
}
|
|
193
|
+
const o = runScenario(dir, opts, root);
|
|
194
|
+
if (o.score === void 0) {
|
|
195
|
+
errored++;
|
|
196
|
+
console.log(`ERROR ${o.name} ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
|
|
197
|
+
} else if (o.pass) {
|
|
198
|
+
passed++;
|
|
199
|
+
console.log(`PASS ${o.name} score=${o.score.toFixed(4)}`);
|
|
200
|
+
} else {
|
|
201
|
+
failed++;
|
|
202
|
+
console.log(`FAIL ${o.name} score=${o.score.toFixed(4)}`);
|
|
203
|
+
}
|
|
204
|
+
}
|
|
205
|
+
console.log(`\nsweep: ${passed} passed, ${failed} failed, ${errored} errored, ${skipped} skipped`);
|
|
206
|
+
process.exit(errored > 0 ? 2 : failed > 0 ? 1 : 0);
|
|
207
|
+
}
|
|
208
|
+
function reduceResults(dir, allowMixed) {
|
|
209
|
+
const entries = [];
|
|
210
|
+
const skipped = [];
|
|
211
|
+
const shas = /* @__PURE__ */ new Set();
|
|
212
|
+
for (const f of fs.readdirSync(dir).filter((f) => f.endsWith(".json") && !f.endsWith(".meta.json")).sort()) {
|
|
213
|
+
let raw;
|
|
214
|
+
try {
|
|
215
|
+
raw = JSON.parse(fs.readFileSync(path.join(dir, f), "utf8"));
|
|
216
|
+
} catch {
|
|
217
|
+
raw = void 0;
|
|
218
|
+
}
|
|
219
|
+
const res = raw?.results?.results?.[0];
|
|
220
|
+
if (typeof res?.score !== "number" || typeof res?.success !== "boolean") {
|
|
221
|
+
console.error(`skipping ${f}: not a promptfoo result`);
|
|
222
|
+
skipped.push(f);
|
|
223
|
+
continue;
|
|
224
|
+
}
|
|
225
|
+
const provider = raw.config?.providers?.[0];
|
|
226
|
+
const judge = raw.config?.defaultTest?.options?.provider;
|
|
227
|
+
const base = f.replace(/\.json$/, "");
|
|
228
|
+
const harness = base.endsWith("--codex") ? "codex" : "claude";
|
|
229
|
+
const [skill, ...rest] = base.replace(/--codex$/, "").split("--");
|
|
230
|
+
let sha = "unattested";
|
|
231
|
+
try {
|
|
232
|
+
sha = JSON.parse(fs.readFileSync(path.join(dir, `${base}.meta.json`), "utf8")).skills_tree_sha ?? "unattested";
|
|
233
|
+
} catch {}
|
|
234
|
+
shas.add(sha);
|
|
235
|
+
entries.push({
|
|
236
|
+
skill,
|
|
237
|
+
scenario: rest.join("--"),
|
|
238
|
+
harness,
|
|
239
|
+
skills_tree_sha: sha,
|
|
240
|
+
score: res.score,
|
|
241
|
+
pass: res.success,
|
|
242
|
+
agent_model: provider?.config?.model ?? "codex-default",
|
|
243
|
+
judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? "unknown",
|
|
244
|
+
latency_ms: res.latencyMs,
|
|
245
|
+
tokens: (res.tokenUsage?.total ?? 0) + (res.tokenUsage?.assertions?.total ?? 0)
|
|
246
|
+
});
|
|
247
|
+
}
|
|
248
|
+
if (shas.size > 1 && !allowMixed) throw new Error(`results span multiple skills-tree revisions (${[...shas].join(", ")}); rerun stale ones or pass --allow-mixed`);
|
|
249
|
+
return {
|
|
250
|
+
treeSha: shas.size === 1 ? [...shas][0] : shas.size === 0 ? "none" : "mixed",
|
|
251
|
+
entries,
|
|
252
|
+
skipped
|
|
253
|
+
};
|
|
254
|
+
}
|
|
255
|
+
function entryKey(e) {
|
|
256
|
+
return [
|
|
257
|
+
e.skill,
|
|
258
|
+
e.scenario,
|
|
259
|
+
e.harness
|
|
260
|
+
].join("\0");
|
|
261
|
+
}
|
|
262
|
+
function mergeScorecard(existing, fresh) {
|
|
263
|
+
const byKey = /* @__PURE__ */ new Map();
|
|
264
|
+
for (const e of existing) byKey.set(entryKey(e), e);
|
|
265
|
+
let carried = byKey.size;
|
|
266
|
+
for (const e of fresh) {
|
|
267
|
+
if (byKey.has(entryKey(e))) carried--;
|
|
268
|
+
byKey.set(entryKey(e), e);
|
|
269
|
+
}
|
|
270
|
+
return {
|
|
271
|
+
entries: [...byKey.values()].sort((a, b) => entryKey(a) < entryKey(b) ? -1 : entryKey(a) > entryKey(b) ? 1 : 0),
|
|
272
|
+
carried
|
|
273
|
+
};
|
|
274
|
+
}
|
|
275
|
+
function treeShaOf(entries) {
|
|
276
|
+
const shas = new Set(entries.map((e) => e.skills_tree_sha));
|
|
277
|
+
return shas.size === 1 ? [...shas][0] : shas.size === 0 ? "none" : "mixed";
|
|
278
|
+
}
|
|
279
|
+
function readExistingScorecard(out) {
|
|
280
|
+
if (!fs.existsSync(out)) return [];
|
|
281
|
+
let prev;
|
|
282
|
+
try {
|
|
283
|
+
prev = JSON.parse(fs.readFileSync(out, "utf8"));
|
|
284
|
+
} catch {
|
|
285
|
+
throw new Error(`existing scorecard ${out} is not valid JSON; refusing to overwrite it`);
|
|
286
|
+
}
|
|
287
|
+
const scenarios = prev?.scenarios;
|
|
288
|
+
if (!Array.isArray(scenarios)) throw new Error(`existing scorecard ${out} has no scenarios array; refusing to overwrite it`);
|
|
289
|
+
return scenarios;
|
|
290
|
+
}
|
|
291
|
+
function cmdSummarize(argv) {
|
|
292
|
+
const { positional, flags } = parseArgs(argv);
|
|
293
|
+
if (positional.length > 0) fail("usage: skillcheck summarize [--root DIR] [--allow-mixed]");
|
|
294
|
+
const dirs = stateDirs(resolveRoot(flags));
|
|
295
|
+
if (!fs.existsSync(dirs.results)) fail(`no results directory at ${dirs.results} — run some evals first`);
|
|
296
|
+
const { entries, skipped } = reduceResults(dirs.results, flags.get("--allow-mixed") === true);
|
|
297
|
+
fs.mkdirSync(dirs.scorecards, { recursive: true });
|
|
298
|
+
const out = path.join(dirs.scorecards, `${(/* @__PURE__ */ new Date()).toISOString().slice(0, 10)}.json`);
|
|
299
|
+
const existing = readExistingScorecard(out);
|
|
300
|
+
const merged = mergeScorecard(existing, entries);
|
|
301
|
+
const scorecard = {
|
|
302
|
+
ran_at: (/* @__PURE__ */ new Date()).toISOString(),
|
|
303
|
+
skills_tree_sha: treeShaOf(merged.entries),
|
|
304
|
+
scenarios: merged.entries
|
|
305
|
+
};
|
|
306
|
+
fs.writeFileSync(out, JSON.stringify(scorecard, null, 2) + "\n");
|
|
307
|
+
console.log(`${out}: ${merged.entries.length} scenario(s), ${merged.entries.filter((e) => e.pass).length} passing, ${skipped.length} skipped file(s)`);
|
|
308
|
+
if (existing.length > 0) console.log(`merged into today's scorecard: ${entries.length} from this run, ${merged.carried} carried over`);
|
|
309
|
+
}
|
|
310
|
+
function cmdLint(argv) {
|
|
311
|
+
const { positional, flags } = parseArgs(argv);
|
|
312
|
+
if (positional.length > 1) fail("usage: skillcheck lint [<root>] [--root DIR]");
|
|
313
|
+
const root = positional.length === 1 ? path.resolve(positional[0]) : resolveRoot(flags);
|
|
314
|
+
const { errors, count } = lintSkills(root);
|
|
315
|
+
if (errors.length > 0) {
|
|
316
|
+
for (const e of errors) console.error(e);
|
|
317
|
+
console.error(`skill lint: ${errors.length} error(s) across ${count} package(s)`);
|
|
318
|
+
process.exit(1);
|
|
319
|
+
}
|
|
320
|
+
console.log(`skill lint: ${count} package(s) clean`);
|
|
321
|
+
}
|
|
322
|
+
function isMainModule() {
|
|
323
|
+
const entry = process.argv[1];
|
|
324
|
+
if (entry === void 0) return false;
|
|
325
|
+
try {
|
|
326
|
+
return fs.realpathSync(entry) === fs.realpathSync(fileURLToPath(import.meta.url));
|
|
327
|
+
} catch {
|
|
328
|
+
return false;
|
|
329
|
+
}
|
|
330
|
+
}
|
|
331
|
+
if (isMainModule()) {
|
|
332
|
+
const [cmd, ...rest] = process.argv.slice(2);
|
|
333
|
+
try {
|
|
334
|
+
if (cmd === "run") cmdRun(rest);
|
|
335
|
+
else if (cmd === "sweep") cmdSweep(rest);
|
|
336
|
+
else if (cmd === "summarize") cmdSummarize(rest);
|
|
337
|
+
else if (cmd === "lint") cmdLint(rest);
|
|
338
|
+
else fail("usage: skillcheck <lint|run|sweep|summarize> ...");
|
|
339
|
+
} catch (err) {
|
|
340
|
+
fail(err instanceof Error ? err.message : String(err));
|
|
341
|
+
}
|
|
342
|
+
}
|
|
343
|
+
//#endregion
|
|
344
|
+
export { classifyResult, mergeScorecard, parseArgs, parseMaxTurns, reduceResults, resolveRoot, stateDirs, toolVersion, treeShaOf };
|
package/dist/lint.js
ADDED
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
import fs from "node:fs";
|
|
2
|
+
import path from "node:path";
|
|
3
|
+
//#region src/lint.ts
|
|
4
|
+
const ALLOWED_KEYS = /* @__PURE__ */ new Set([
|
|
5
|
+
"name",
|
|
6
|
+
"description",
|
|
7
|
+
"disable-model-invocation"
|
|
8
|
+
]);
|
|
9
|
+
function lintSkill(dir, root, errors) {
|
|
10
|
+
const skillMd = path.join(dir, "SKILL.md");
|
|
11
|
+
const label = path.relative(root, skillMd);
|
|
12
|
+
if (!fs.existsSync(skillMd)) {
|
|
13
|
+
errors.push(`${path.relative(root, dir)}: missing SKILL.md`);
|
|
14
|
+
return;
|
|
15
|
+
}
|
|
16
|
+
const lines = fs.readFileSync(skillMd, "utf8").split("\n");
|
|
17
|
+
if (lines[0] !== "---") {
|
|
18
|
+
errors.push(`${label}: frontmatter must open with --- on line 1`);
|
|
19
|
+
return;
|
|
20
|
+
}
|
|
21
|
+
const close = lines.indexOf("---", 1);
|
|
22
|
+
if (close === -1) {
|
|
23
|
+
errors.push(`${label}: frontmatter never closes`);
|
|
24
|
+
return;
|
|
25
|
+
}
|
|
26
|
+
const fields = /* @__PURE__ */ new Map();
|
|
27
|
+
for (const line of lines.slice(1, close)) {
|
|
28
|
+
const m = line.match(/^([a-z-]+):\s*(.*)$/);
|
|
29
|
+
if (!m) {
|
|
30
|
+
errors.push(`${label}: unparseable frontmatter line: ${line}`);
|
|
31
|
+
continue;
|
|
32
|
+
}
|
|
33
|
+
const [, key, raw] = m;
|
|
34
|
+
if (!ALLOWED_KEYS.has(key)) {
|
|
35
|
+
errors.push(`${label}: unknown frontmatter key: ${key}`);
|
|
36
|
+
continue;
|
|
37
|
+
}
|
|
38
|
+
if (fields.has(key)) {
|
|
39
|
+
errors.push(`${label}: duplicate frontmatter key: ${key}`);
|
|
40
|
+
continue;
|
|
41
|
+
}
|
|
42
|
+
const value = raw.replace(/^"(.*)"$/s, "$1").replace(/^'(.*)'$/s, "$1").trim();
|
|
43
|
+
fields.set(key, {
|
|
44
|
+
raw: raw.trim(),
|
|
45
|
+
value
|
|
46
|
+
});
|
|
47
|
+
}
|
|
48
|
+
const name = fields.get("name")?.value ?? "";
|
|
49
|
+
if (name !== path.basename(dir)) errors.push(`${label}: frontmatter name ${JSON.stringify(name)} != directory ${path.basename(dir)}`);
|
|
50
|
+
if (!fields.get("description")?.value) errors.push(`${label}: description is required and must be non-empty`);
|
|
51
|
+
const dmi = fields.get("disable-model-invocation");
|
|
52
|
+
if (dmi !== void 0 && dmi.raw !== "true") errors.push(`${label}: disable-model-invocation must be the literal boolean true, got ${JSON.stringify(dmi.raw)}`);
|
|
53
|
+
const body = lines.slice(close + 1).join("\n").replace(/^```[\s\S]*?^```/gm, "").replace(/`[^`\n]*`/g, "");
|
|
54
|
+
for (const link of body.matchAll(/\]\(\s*(<[^>\n]*>|[^)\s]+)(?:\s+"[^"\n]*")?\s*\)/g)) {
|
|
55
|
+
const target = link[1].replace(/^<(.*)>$/, "$1");
|
|
56
|
+
if (/^[a-z][a-z+.-]*:/.test(target) || target.startsWith("#") || target === "") continue;
|
|
57
|
+
const resolved = path.join(dir, target.split("#")[0]);
|
|
58
|
+
if (!fs.existsSync(resolved)) errors.push(`${label}: link target does not exist: ${target}`);
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
function lintSkills(root) {
|
|
62
|
+
const roots = ["skills", ...fs.globSync("cli/*/skills", { cwd: root })].map((r) => path.join(root, r));
|
|
63
|
+
const errors = [];
|
|
64
|
+
let count = 0;
|
|
65
|
+
for (const dir of roots) {
|
|
66
|
+
if (!fs.existsSync(dir)) continue;
|
|
67
|
+
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
|
68
|
+
if (!entry.isDirectory() || entry.name.startsWith(".")) continue;
|
|
69
|
+
lintSkill(path.join(dir, entry.name), root, errors);
|
|
70
|
+
count++;
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
return {
|
|
74
|
+
errors,
|
|
75
|
+
count
|
|
76
|
+
};
|
|
77
|
+
}
|
|
78
|
+
//#endregion
|
|
79
|
+
export { lintSkills };
|
package/dist/scenario.js
ADDED
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
import { createRequire } from "node:module";
|
|
2
|
+
import fs from "node:fs";
|
|
3
|
+
import path from "node:path";
|
|
4
|
+
import { createHash } from "node:crypto";
|
|
5
|
+
import os from "node:os";
|
|
6
|
+
//#region src/scenario.ts
|
|
7
|
+
function loadScenario(scenarioDir) {
|
|
8
|
+
const match = scenarioDir.match(/skills\/([^/]+)\/evals\/([^/]+)$/);
|
|
9
|
+
if (!match) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
|
|
10
|
+
const [, skill, scenario] = match;
|
|
11
|
+
const taskMd = fs.readFileSync(path.join(scenarioDir, "task.md"), "utf8");
|
|
12
|
+
const criteria = JSON.parse(fs.readFileSync(path.join(scenarioDir, "criteria.json"), "utf8"));
|
|
13
|
+
if (criteria.type !== "weighted_checklist" || !Array.isArray(criteria.checklist) || criteria.checklist.length === 0) throw new Error(`unsupported or empty criteria in ${scenarioDir}`);
|
|
14
|
+
for (const item of criteria.checklist) if (!(typeof item?.name === "string" && item.name.trim() !== "" && typeof item?.description === "string" && item.description.trim() !== "" && Number.isFinite(item?.max_score) && item.max_score > 0)) throw new Error(`invalid checklist item in ${scenarioDir}: ${JSON.stringify(item)}`);
|
|
15
|
+
const files = [];
|
|
16
|
+
let prompt = taskMd.replace(/^=+ FILE: (.+?) =+\n([\s\S]*?)\n=+ END FILE =+$/gm, (_, name, content) => {
|
|
17
|
+
files.push({
|
|
18
|
+
name: name.trim(),
|
|
19
|
+
content: content + "\n"
|
|
20
|
+
});
|
|
21
|
+
return `(Input file \`${name.trim()}\` is available in your working directory.)`;
|
|
22
|
+
});
|
|
23
|
+
const skillDir = path.resolve(scenarioDir, "../..");
|
|
24
|
+
if (isHiddenSkill(fs.readFileSync(path.join(skillDir, "SKILL.md"), "utf8"))) prompt = `Use the ${skill} skill for this task.\n\n${prompt}`;
|
|
25
|
+
return {
|
|
26
|
+
skill,
|
|
27
|
+
scenario,
|
|
28
|
+
name: `${skill}--${scenario}`,
|
|
29
|
+
skillDir,
|
|
30
|
+
prompt,
|
|
31
|
+
files,
|
|
32
|
+
criteria
|
|
33
|
+
};
|
|
34
|
+
}
|
|
35
|
+
function runNameFor(scenarioDir, harness) {
|
|
36
|
+
const m = path.resolve(scenarioDir).match(/skills\/([^/]+)\/evals\/([^/]+)$/);
|
|
37
|
+
if (!m) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
|
|
38
|
+
return harness === "codex" ? `${m[1]}--${m[2]}--codex` : `${m[1]}--${m[2]}`;
|
|
39
|
+
}
|
|
40
|
+
function frontmatterRange(text) {
|
|
41
|
+
const lines = text.split("\n");
|
|
42
|
+
if (lines[0] !== "---") return null;
|
|
43
|
+
const close = lines.indexOf("---", 1);
|
|
44
|
+
return close === -1 ? null : [1, close];
|
|
45
|
+
}
|
|
46
|
+
function isHiddenSkill(skillMd) {
|
|
47
|
+
const range = frontmatterRange(skillMd);
|
|
48
|
+
if (!range) return false;
|
|
49
|
+
return skillMd.split("\n").slice(range[0], range[1]).some((l) => l.startsWith("disable-model-invocation:"));
|
|
50
|
+
}
|
|
51
|
+
function stripHiddenFlag(skillMd) {
|
|
52
|
+
const range = frontmatterRange(skillMd);
|
|
53
|
+
if (!range) return skillMd;
|
|
54
|
+
return skillMd.split("\n").filter((l, i) => !(i >= range[0] && i < range[1] && l.startsWith("disable-model-invocation:"))).join("\n");
|
|
55
|
+
}
|
|
56
|
+
const RESERVED = /* @__PURE__ */ new Set([".claude", ".agents"]);
|
|
57
|
+
function materialize(s, runDir, harness) {
|
|
58
|
+
const workdir = path.join(runDir, "workdir");
|
|
59
|
+
fs.rmSync(runDir, {
|
|
60
|
+
recursive: true,
|
|
61
|
+
force: true
|
|
62
|
+
});
|
|
63
|
+
fs.mkdirSync(workdir, { recursive: true });
|
|
64
|
+
const seen = /* @__PURE__ */ new Set();
|
|
65
|
+
const planned = s.files.map((f) => {
|
|
66
|
+
const dest = path.resolve(workdir, f.name);
|
|
67
|
+
const rel = path.relative(workdir, dest);
|
|
68
|
+
if (rel === "" || rel.startsWith("..") || path.isAbsolute(rel)) throw new Error(`embedded file escapes workdir: ${f.name}`);
|
|
69
|
+
if (RESERVED.has(rel.split(path.sep)[0])) throw new Error(`embedded file targets reserved dir: ${f.name}`);
|
|
70
|
+
if (seen.has(dest)) throw new Error(`duplicate embedded file: ${f.name}`);
|
|
71
|
+
seen.add(dest);
|
|
72
|
+
return {
|
|
73
|
+
dest,
|
|
74
|
+
content: f.content
|
|
75
|
+
};
|
|
76
|
+
});
|
|
77
|
+
for (const { dest, content } of planned) {
|
|
78
|
+
fs.mkdirSync(path.dirname(dest), { recursive: true });
|
|
79
|
+
fs.writeFileSync(dest, content);
|
|
80
|
+
}
|
|
81
|
+
const roots = harness === "codex" ? [".claude", ".agents"] : [".claude"];
|
|
82
|
+
for (const root of roots) fs.cpSync(s.skillDir, path.join(workdir, root, "skills", s.skill), {
|
|
83
|
+
recursive: true,
|
|
84
|
+
filter: (src) => path.basename(src) !== "evals"
|
|
85
|
+
});
|
|
86
|
+
for (const root of roots) {
|
|
87
|
+
const skillMd = path.join(workdir, root, "skills", s.skill, "SKILL.md");
|
|
88
|
+
const text = fs.readFileSync(skillMd, "utf8");
|
|
89
|
+
const stripped = stripHiddenFlag(text);
|
|
90
|
+
if (stripped !== text) fs.writeFileSync(skillMd, stripped);
|
|
91
|
+
}
|
|
92
|
+
const manifest = {};
|
|
93
|
+
const walk = (dir) => {
|
|
94
|
+
for (const e of fs.readdirSync(dir, { withFileTypes: true })) {
|
|
95
|
+
const p = path.join(dir, e.name);
|
|
96
|
+
if (e.isDirectory()) {
|
|
97
|
+
if (e.name !== ".claude") walk(p);
|
|
98
|
+
} else manifest[path.relative(workdir, p)] = createHash("sha256").update(fs.readFileSync(p)).digest("hex");
|
|
99
|
+
}
|
|
100
|
+
};
|
|
101
|
+
walk(workdir);
|
|
102
|
+
const manifestPath = path.join(runDir, "manifest.json");
|
|
103
|
+
fs.writeFileSync(manifestPath, JSON.stringify(manifest, null, 2));
|
|
104
|
+
return {
|
|
105
|
+
workdir,
|
|
106
|
+
manifestPath
|
|
107
|
+
};
|
|
108
|
+
}
|
|
109
|
+
function agentProvider(opts, workdir, skill) {
|
|
110
|
+
if (opts.harness === "codex") return {
|
|
111
|
+
id: "openai:codex-sdk",
|
|
112
|
+
config: {
|
|
113
|
+
...opts.agentModel ? { model: opts.agentModel } : {},
|
|
114
|
+
working_dir: workdir,
|
|
115
|
+
skip_git_repo_check: true,
|
|
116
|
+
enable_streaming: true,
|
|
117
|
+
sandbox_mode: "workspace-write",
|
|
118
|
+
cli_env: { CODEX_HOME: process.env.CODEX_HOME ?? path.join(os.homedir(), ".codex") }
|
|
119
|
+
}
|
|
120
|
+
};
|
|
121
|
+
return {
|
|
122
|
+
id: "anthropic:claude-agent-sdk",
|
|
123
|
+
config: {
|
|
124
|
+
model: opts.agentModel ?? "claude-opus-5",
|
|
125
|
+
apiKeyRequired: false,
|
|
126
|
+
working_dir: workdir,
|
|
127
|
+
setting_sources: ["project"],
|
|
128
|
+
skills: [skill],
|
|
129
|
+
permission_mode: "acceptEdits",
|
|
130
|
+
append_allowed_tools: [
|
|
131
|
+
"Read",
|
|
132
|
+
"Write",
|
|
133
|
+
"Edit",
|
|
134
|
+
"Glob",
|
|
135
|
+
"Grep"
|
|
136
|
+
],
|
|
137
|
+
max_turns: opts.maxTurns ?? 50
|
|
138
|
+
}
|
|
139
|
+
};
|
|
140
|
+
}
|
|
141
|
+
function buildConfig(s, workdir, manifestPath, opts, transformPath) {
|
|
142
|
+
return {
|
|
143
|
+
description: `${s.skill}/${s.scenario}`,
|
|
144
|
+
prompts: ["{{task}}"],
|
|
145
|
+
providers: [agentProvider(opts, workdir, s.skill)],
|
|
146
|
+
defaultTest: { options: {
|
|
147
|
+
provider: process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
|
|
148
|
+
id: "anthropic:claude-agent-sdk",
|
|
149
|
+
config: {
|
|
150
|
+
model: opts.judgeModel,
|
|
151
|
+
apiKeyRequired: false,
|
|
152
|
+
max_turns: 3,
|
|
153
|
+
output_format: {
|
|
154
|
+
type: "json_schema",
|
|
155
|
+
schema: {
|
|
156
|
+
type: "object",
|
|
157
|
+
additionalProperties: false,
|
|
158
|
+
required: [
|
|
159
|
+
"reason",
|
|
160
|
+
"pass",
|
|
161
|
+
"score"
|
|
162
|
+
],
|
|
163
|
+
properties: {
|
|
164
|
+
reason: { type: "string" },
|
|
165
|
+
pass: { type: "boolean" },
|
|
166
|
+
score: {
|
|
167
|
+
type: "number",
|
|
168
|
+
minimum: 0,
|
|
169
|
+
maximum: 1
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
},
|
|
176
|
+
transform: `file://${transformPath}`
|
|
177
|
+
} },
|
|
178
|
+
tests: [{
|
|
179
|
+
description: s.criteria.context,
|
|
180
|
+
vars: {
|
|
181
|
+
task: s.prompt,
|
|
182
|
+
workdir,
|
|
183
|
+
manifest: manifestPath
|
|
184
|
+
},
|
|
185
|
+
assert: [{
|
|
186
|
+
type: "assert-set",
|
|
187
|
+
threshold: .7,
|
|
188
|
+
assert: s.criteria.checklist.map((item) => ({
|
|
189
|
+
type: "llm-rubric",
|
|
190
|
+
value: `${item.name}: ${item.description}`,
|
|
191
|
+
weight: item.max_score
|
|
192
|
+
}))
|
|
193
|
+
}, {
|
|
194
|
+
type: "skill-used",
|
|
195
|
+
value: s.skill
|
|
196
|
+
}]
|
|
197
|
+
}]
|
|
198
|
+
};
|
|
199
|
+
}
|
|
200
|
+
const SDK_PACKAGES = ["@anthropic-ai/claude-agent-sdk", "@openai/codex-sdk"];
|
|
201
|
+
function sdkNodeModulesDir() {
|
|
202
|
+
const require = createRequire(import.meta.url);
|
|
203
|
+
for (const pkg of SDK_PACKAGES) for (const dir of require.resolve.paths(pkg) ?? []) if (fs.existsSync(path.join(dir, pkg, "package.json"))) return dir;
|
|
204
|
+
}
|
|
205
|
+
function generateRun(scenarioDir, opts, paths) {
|
|
206
|
+
const s = loadScenario(scenarioDir);
|
|
207
|
+
const name = runNameFor(scenarioDir, opts.harness);
|
|
208
|
+
const runDir = path.join(paths.scratchDir, name);
|
|
209
|
+
const { workdir, manifestPath } = materialize(s, runDir, opts.harness);
|
|
210
|
+
const sdkDir = sdkNodeModulesDir();
|
|
211
|
+
if (sdkDir !== void 0) {
|
|
212
|
+
const link = path.join(runDir, "node_modules");
|
|
213
|
+
fs.rmSync(link, {
|
|
214
|
+
recursive: true,
|
|
215
|
+
force: true
|
|
216
|
+
});
|
|
217
|
+
fs.symlinkSync(sdkDir, link, "dir");
|
|
218
|
+
}
|
|
219
|
+
const config = buildConfig(s, workdir, manifestPath, opts, paths.transformPath);
|
|
220
|
+
const configPath = path.join(runDir, "promptfooconfig.json");
|
|
221
|
+
fs.writeFileSync(configPath, JSON.stringify(config, null, 2));
|
|
222
|
+
return {
|
|
223
|
+
name,
|
|
224
|
+
configPath
|
|
225
|
+
};
|
|
226
|
+
}
|
|
227
|
+
//#endregion
|
|
228
|
+
export { buildConfig, generateRun, isHiddenSkill, loadScenario, materialize, runNameFor, sdkNodeModulesDir, stripHiddenFlag };
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import fs from "node:fs";
|
|
2
|
+
import path from "node:path";
|
|
3
|
+
import { createHash } from "node:crypto";
|
|
4
|
+
//#region src/transform.ts
|
|
5
|
+
const PER_FILE_CAP = 4e3;
|
|
6
|
+
const TOTAL_CAP = 24e3;
|
|
7
|
+
function safeSlice(text, end) {
|
|
8
|
+
let cut = text.slice(0, end);
|
|
9
|
+
const last = cut.charCodeAt(cut.length - 1);
|
|
10
|
+
if (last >= 55296 && last <= 56319) cut = cut.slice(0, -1);
|
|
11
|
+
return cut;
|
|
12
|
+
}
|
|
13
|
+
function capped(rel, text) {
|
|
14
|
+
if (text.length <= PER_FILE_CAP) return `=== OUTPUT FILE: ${rel} ===\n${text}\n=== END OUTPUT FILE ===`;
|
|
15
|
+
return `=== OUTPUT FILE: ${rel} (truncated: showing ${PER_FILE_CAP} of ${text.length} chars) ===\n${safeSlice(text, PER_FILE_CAP)}\n=== END OUTPUT FILE ===`;
|
|
16
|
+
}
|
|
17
|
+
function transform(output, context) {
|
|
18
|
+
const workdir = context.vars.workdir;
|
|
19
|
+
const manifest = JSON.parse(fs.readFileSync(context.vars.manifest, "utf8"));
|
|
20
|
+
const sections = [];
|
|
21
|
+
const visited = /* @__PURE__ */ new Set();
|
|
22
|
+
const walk = (dir) => {
|
|
23
|
+
for (const e of fs.readdirSync(dir, { withFileTypes: true })) {
|
|
24
|
+
const p = path.join(dir, e.name);
|
|
25
|
+
if (e.isDirectory()) {
|
|
26
|
+
if (e.name !== ".claude") walk(p);
|
|
27
|
+
} else if (e.isFile()) {
|
|
28
|
+
const rel = path.relative(workdir, p);
|
|
29
|
+
visited.add(rel);
|
|
30
|
+
let body;
|
|
31
|
+
try {
|
|
32
|
+
body = fs.readFileSync(p);
|
|
33
|
+
} catch (err) {
|
|
34
|
+
const detail = err instanceof Error ? err.message : String(err);
|
|
35
|
+
sections.push({
|
|
36
|
+
rel,
|
|
37
|
+
text: `=== UNREADABLE FILE: ${rel} === (${detail})`
|
|
38
|
+
});
|
|
39
|
+
continue;
|
|
40
|
+
}
|
|
41
|
+
const hash = createHash("sha256").update(body).digest("hex");
|
|
42
|
+
if (manifest[rel] !== hash) sections.push({
|
|
43
|
+
rel,
|
|
44
|
+
text: capped(rel, body.toString("utf8"))
|
|
45
|
+
});
|
|
46
|
+
} else {
|
|
47
|
+
const nrel = path.relative(workdir, p);
|
|
48
|
+
sections.push({
|
|
49
|
+
rel: nrel,
|
|
50
|
+
text: `=== NON-REGULAR FILE: ${nrel} ===`
|
|
51
|
+
});
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
};
|
|
55
|
+
walk(workdir);
|
|
56
|
+
for (const rel of Object.keys(manifest)) if (!visited.has(rel)) sections.push({
|
|
57
|
+
rel,
|
|
58
|
+
text: `=== DELETED FILE: ${rel} === (input file removed by the agent)`
|
|
59
|
+
});
|
|
60
|
+
if (sections.length === 0) return output;
|
|
61
|
+
sections.sort((a, b) => a.rel < b.rel ? -1 : a.rel > b.rel ? 1 : 0);
|
|
62
|
+
let appended = sections.map((s) => s.text).join("\n\n");
|
|
63
|
+
if (appended.length > TOTAL_CAP) appended = `${safeSlice(appended, TOTAL_CAP)}\n\n=== TRUNCATED: output files exceeded ${TOTAL_CAP} chars total ===`;
|
|
64
|
+
return `${output}\n\n${appended}`;
|
|
65
|
+
}
|
|
66
|
+
//#endregion
|
|
67
|
+
export { transform as default };
|
package/docs/adoption.md
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
# adopting it in a repo
|
|
2
|
+
|
|
3
|
+
two separate decisions: run the lint in CI, and run evals on a machine that has
|
|
4
|
+
model auth. only the first belongs in a consumer repo.
|
|
5
|
+
|
|
6
|
+
## lint in CI
|
|
7
|
+
|
|
8
|
+
```sh
|
|
9
|
+
pnpm add -D @uinaf/skillcheck
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
runners that install with `--ignore-scripts` are fine: the package ships
|
|
13
|
+
compiled ESM and has no install, prepare, or postinstall script.
|
|
14
|
+
|
|
15
|
+
```json
|
|
16
|
+
{ "scripts": { "skills:lint": "skillcheck lint" } }
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
```yaml
|
|
20
|
+
jobs:
|
|
21
|
+
skills:
|
|
22
|
+
name: skills lint
|
|
23
|
+
runs-on: ubuntu-latest
|
|
24
|
+
steps:
|
|
25
|
+
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
26
|
+
with:
|
|
27
|
+
persist-credentials: false
|
|
28
|
+
- uses: actions/setup-node@249970729cb0ef3589644e2896645e5dc5ba9c38 # v6.5.0
|
|
29
|
+
with:
|
|
30
|
+
node-version: "24"
|
|
31
|
+
- run: pnpm install --frozen-lockfile
|
|
32
|
+
- run: pnpm run skills:lint
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
the job needs no secrets and no network beyond the install. node 24 is the
|
|
36
|
+
floor.
|
|
37
|
+
|
|
38
|
+
run it through the script rather than a bare `npx skillcheck`: the script
|
|
39
|
+
resolves the version the repo pinned, and `npx` would resolve the latest one on
|
|
40
|
+
the registry.
|
|
41
|
+
|
|
42
|
+
## evals
|
|
43
|
+
|
|
44
|
+
sweeps need model credentials, so they stay off consumer CI and run from an
|
|
45
|
+
operator machine or a job that already holds gateway auth:
|
|
46
|
+
|
|
47
|
+
```sh
|
|
48
|
+
skillcheck sweep # resumes: only scenarios without results
|
|
49
|
+
skillcheck summarize # writes .skillcheck/scorecards/<UTC-date>.json
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
commit `.skillcheck/scorecards/`. gitignore the rest:
|
|
53
|
+
|
|
54
|
+
```gitignore
|
|
55
|
+
.skillcheck/results/
|
|
56
|
+
.skillcheck/scratch/
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
a scorecard is only comparable against the tree it graded, which is why every
|
|
60
|
+
result carries the root repo's HEAD and `summarize` refuses to mix revisions
|
|
61
|
+
without `--allow-mixed`.
|
|
62
|
+
|
|
63
|
+
## upgrading
|
|
64
|
+
|
|
65
|
+
bump the version in `package.json` and rerun the sweep. results carry the
|
|
66
|
+
`tool_version` that produced them, so a scorecard says which harness build it
|
|
67
|
+
came from as well as which skills tree.
|
|
68
|
+
|
|
69
|
+
## the older install path
|
|
70
|
+
|
|
71
|
+
before the package was published, consumers installed it from a git tag:
|
|
72
|
+
|
|
73
|
+
```sh
|
|
74
|
+
npm i -D github:uinaf/skillcheck#v0.1.3
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
those tags are frozen and still work: they carry a committed `dist/`, and npm
|
|
78
|
+
12 consumers needed `allow-git=root` in `.npmrc` to accept the spec at all. the
|
|
79
|
+
registry install needs none of that. tags from `v0.1.4` on are npm releases and
|
|
80
|
+
carry no `dist/`, so a git spec pointing at one will not run.
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
# releasing
|
|
2
|
+
|
|
3
|
+
## pipeline
|
|
4
|
+
|
|
5
|
+
a push to `main` runs one workflow, `.github/workflows/release.yml`:
|
|
6
|
+
|
|
7
|
+
```text
|
|
8
|
+
verify ──┐
|
|
9
|
+
├──> release npm publish, OIDC + uinaf-releaser (release environment)
|
|
10
|
+
secrets ─┘
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
`verify` and `secrets` are the shared gate, called from `verify.yml` and
|
|
14
|
+
`secrets.yml`, so one definition serves pull requests, the merge queue, and this
|
|
15
|
+
push. keep it that way: a second copy of the gate on a push-to-`main` workflow
|
|
16
|
+
races this one over the same commit.
|
|
17
|
+
|
|
18
|
+
the file name `release.yml` is load-bearing. see below.
|
|
19
|
+
|
|
20
|
+
## npm
|
|
21
|
+
|
|
22
|
+
`@uinaf/skillcheck` publishes from `.github/workflows/release.yml` via npm
|
|
23
|
+
Trusted Publishing (OIDC) and the `uinaf-releaser` GitHub App. there is no npm
|
|
24
|
+
token in this repository, in its environments, or in the organization.
|
|
25
|
+
|
|
26
|
+
required on the `release` GitHub Environment:
|
|
27
|
+
|
|
28
|
+
| name | kind | purpose |
|
|
29
|
+
| ------------------------------- | ------ | ------------------------------------------- |
|
|
30
|
+
| `UINAF_RELEASE_APP_CLIENT_ID` | var | GitHub App client id for the releaser bot |
|
|
31
|
+
| `UINAF_RELEASE_APP_PRIVATE_KEY` | secret | GitHub App private key for the releaser bot |
|
|
32
|
+
|
|
33
|
+
the trusted publisher on npmjs.com is registered by **file path**, so
|
|
34
|
+
`.github/workflows/release.yml` cannot be renamed or moved without editing that
|
|
35
|
+
registration first. a rename fails the publish with an identity mismatch, and
|
|
36
|
+
nothing earlier in the run reports it. the `release` environment name is bound
|
|
37
|
+
the same way.
|
|
38
|
+
|
|
39
|
+
deleting the `release` environment deletes both rows above with it, and there is
|
|
40
|
+
no repo-level fallback: `create-github-app-token` then runs with empty inputs
|
|
41
|
+
and the job fails at that step. the private key cannot be read back from
|
|
42
|
+
GitHub; recreating it means generating a new one in the App settings.
|
|
43
|
+
|
|
44
|
+
## version history
|
|
45
|
+
|
|
46
|
+
semantic-release owns the version and the tag. `tagFormat` is `v${version}` and
|
|
47
|
+
history continues from `v0.1.3`; `v0.1.0`–`v0.1.3` are the legacy git-install
|
|
48
|
+
tags and are never deleted or moved.
|
|
49
|
+
|
|
50
|
+
during preparation, `@semantic-release/npm` stages the released `package.json`
|
|
51
|
+
version and `@jno21/semantic-release-github-commit` commits it to `main` through
|
|
52
|
+
GitHub's API as the authenticated App. GitHub signs that commit, and the release
|
|
53
|
+
tag points to it. the `[skip ci]` marker on that commit is what stops a release
|
|
54
|
+
from releasing itself.
|
|
55
|
+
|
|
56
|
+
check what the next version would be without publishing anything:
|
|
57
|
+
|
|
58
|
+
```sh
|
|
59
|
+
pnpm dlx semantic-release --dry-run --no-ci
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
## the artifact
|
|
63
|
+
|
|
64
|
+
`dist/` is generated and untracked. `prepublishOnly` runs `pnpm run verify`,
|
|
65
|
+
which builds it, so the tarball is always packed from a tree that just passed
|
|
66
|
+
the gate. `files` is `dist`, `docs`, `README.md`, `LICENSE`.
|
|
67
|
+
|
|
68
|
+
manual publish is emergency recovery only:
|
|
69
|
+
|
|
70
|
+
```sh
|
|
71
|
+
pnpm run verify
|
|
72
|
+
npm publish --access public
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
## the old install path
|
|
76
|
+
|
|
77
|
+
before `@uinaf/skillcheck` existed, consumers installed
|
|
78
|
+
`github:uinaf/skillcheck#v0.1.3`. those tags still resolve and still carry a
|
|
79
|
+
committed `dist/`, so anything pinned to them keeps working untouched. new
|
|
80
|
+
consumers use npm.
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
# writing scenarios
|
|
2
|
+
|
|
3
|
+
a scenario is two files in a frozen location:
|
|
4
|
+
|
|
5
|
+
```text
|
|
6
|
+
<root>/skills/<skill>/evals/<scenario>/task.md
|
|
7
|
+
<root>/skills/<skill>/evals/<scenario>/criteria.json
|
|
8
|
+
```
|
|
9
|
+
|
|
10
|
+
the path is the identity: `<skill>--<scenario>` names the run, the result file,
|
|
11
|
+
and the scorecard entry. on the codex harness the name gains a `--codex` suffix,
|
|
12
|
+
so both harnesses can hold results side by side. a directory missing either file
|
|
13
|
+
is not discovered.
|
|
14
|
+
|
|
15
|
+
## task.md
|
|
16
|
+
|
|
17
|
+
the prompt handed to the agent, verbatim, with one piece of syntax. input files
|
|
18
|
+
are embedded inline and materialized into the workdir before the run:
|
|
19
|
+
|
|
20
|
+
```md
|
|
21
|
+
Fix the failing check in the config below.
|
|
22
|
+
|
|
23
|
+
======= FILE: config.json =======
|
|
24
|
+
{ "retries": -1 }
|
|
25
|
+
======= END FILE =======
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
each block is replaced in the prompt with a pointer ("Input file `config.json`
|
|
29
|
+
is available in your working directory.") and written to disk. destinations
|
|
30
|
+
must stay under the workdir, must not collide, and must not target `.claude/` or
|
|
31
|
+
`.agents/`, since a fixture that writes agent config would be configuring its
|
|
32
|
+
own examiner.
|
|
33
|
+
|
|
34
|
+
write the task the way a user would write it. do not name the skill, describe
|
|
35
|
+
its steps, or hint at the checklist: routing is part of what is being measured.
|
|
36
|
+
|
|
37
|
+
## criteria.json
|
|
38
|
+
|
|
39
|
+
```json
|
|
40
|
+
{
|
|
41
|
+
"type": "weighted_checklist",
|
|
42
|
+
"context": "one line describing what a good answer looks like",
|
|
43
|
+
"checklist": [
|
|
44
|
+
{ "name": "short-handle", "description": "what the judge should look for", "max_score": 3 }
|
|
45
|
+
]
|
|
46
|
+
}
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
`type` must be `weighted_checklist` and the checklist must be non-empty. every
|
|
50
|
+
item needs a non-empty `name` and `description` and a positive `max_score`.
|
|
51
|
+
|
|
52
|
+
each item becomes one `llm-rubric` assertion weighted by `max_score`, inside an
|
|
53
|
+
assert-set with threshold 0.7. a separate `skill-used` assertion sits outside
|
|
54
|
+
that aggregate, so a run that produces good output without ever loading the
|
|
55
|
+
skill still fails. there is no test-level threshold: both must pass.
|
|
56
|
+
|
|
57
|
+
write descriptions a judge can check against the deliverable: an observable
|
|
58
|
+
property, not a feeling. weight the items that would make a reviewer reject the
|
|
59
|
+
work.
|
|
60
|
+
|
|
61
|
+
## what the judge sees
|
|
62
|
+
|
|
63
|
+
the agent's final message, plus every file in the workdir that differs from the
|
|
64
|
+
pre-run manifest. unchanged inputs are omitted; deleted inputs, unreadable
|
|
65
|
+
files, and non-regular files are named rather than read.
|
|
66
|
+
|
|
67
|
+
sections are sorted by path, each file is capped at 4,000 characters and the
|
|
68
|
+
appended total at 24,000, with truncation stated inline. very large outputs make
|
|
69
|
+
rubric judges return nothing at all, which is why the caps exist. keep fixtures
|
|
70
|
+
small enough that the deliverable fits.
|
|
71
|
+
|
|
72
|
+
## hidden skills
|
|
73
|
+
|
|
74
|
+
a skill with `disable-model-invocation: true` is explicit-invoke-only in
|
|
75
|
+
production, which the agent SDK cannot simulate. so the eval copy, never the shipped one,
|
|
76
|
+
has the flag stripped, and the task gains a leading
|
|
77
|
+
`Use the <skill> skill for this task.` the eval then measures
|
|
78
|
+
behavior-when-invoked rather than routing. the flag is only honored inside the
|
|
79
|
+
frontmatter block; body text mentioning the key does not count.
|
|
80
|
+
|
|
81
|
+
## the workdir
|
|
82
|
+
|
|
83
|
+
per run, under `<root>/.skillcheck/scratch/<name>/`, rebuilt from scratch each
|
|
84
|
+
time. the skill under test is installed at `.claude/skills/<skill>/` (and also
|
|
85
|
+
`.agents/skills/<skill>/` on the codex harness) with its `evals/` directory
|
|
86
|
+
excluded, so criteria never leak into the agent's context.
|
package/docs/usage.md
ADDED
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
# usage
|
|
2
|
+
|
|
3
|
+
every subcommand resolves one root, `--root <dir>` or the current directory.
|
|
4
|
+
`lint` also takes the root as a positional, because that is the shape CI reaches
|
|
5
|
+
for first.
|
|
6
|
+
|
|
7
|
+
## lint
|
|
8
|
+
|
|
9
|
+
```sh
|
|
10
|
+
skillcheck lint # lints the current repo
|
|
11
|
+
skillcheck lint ../other # lints another root
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
checks each `<root>/skills/<skill>/`:
|
|
15
|
+
|
|
16
|
+
- frontmatter opens with `---` on line 1 and closes
|
|
17
|
+
- keys are `name`, `description`, `disable-model-invocation` and nothing else,
|
|
18
|
+
each at most once
|
|
19
|
+
- `name` equals the directory name; `description` is non-empty
|
|
20
|
+
- `disable-model-invocation`, when present, is the bare YAML boolean `true`.
|
|
21
|
+
a quoted `"true"` is an error
|
|
22
|
+
- relative links in the body resolve on disk
|
|
23
|
+
|
|
24
|
+
code spans and fenced blocks are stripped before links are checked, so example
|
|
25
|
+
links never fail. external schemes and `#anchors` pass. dot-directories under
|
|
26
|
+
`skills/` (`.claude-plugin`) are plugin metadata, not packages, and are skipped.
|
|
27
|
+
|
|
28
|
+
findings print one per line, relative to the linted root, then a count. exit 0
|
|
29
|
+
clean, 1 with findings.
|
|
30
|
+
|
|
31
|
+
## run
|
|
32
|
+
|
|
33
|
+
```sh
|
|
34
|
+
skillcheck run skills/<skill>/evals/<scenario>
|
|
35
|
+
skillcheck run <scenario-dir> --agent MODEL --judge MODEL --harness codex --max-turns 80
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
materializes the scenario into `<root>/.skillcheck/scratch/<name>/workdir`,
|
|
39
|
+
installs the skill under test into that workdir, drives the agent, and grades
|
|
40
|
+
the files it wrote. exit 0 pass, 1 graded fail, 2 error (promptfoo produced no
|
|
41
|
+
usable result).
|
|
42
|
+
|
|
43
|
+
a test that errored was never graded, so it exits 2, prints the provider's
|
|
44
|
+
message, and writes no provenance sidecar. it is never reported as
|
|
45
|
+
`FAIL score=0.0000`; only a real judged verdict can fail a run.
|
|
46
|
+
|
|
47
|
+
defaults: `--harness claude`, agent `claude-opus-5`, judge `claude-opus-5`,
|
|
48
|
+
`--max-turns 50`. on the codex harness, omitting `--agent` leaves the model to
|
|
49
|
+
the Codex CLI's own default.
|
|
50
|
+
|
|
51
|
+
## sweep
|
|
52
|
+
|
|
53
|
+
```sh
|
|
54
|
+
skillcheck sweep # only scenarios without results
|
|
55
|
+
skillcheck sweep --all # rerun everything
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
walks `<root>/skills/*/evals/*` and `<root>/cli/*/skills/*/evals/*`, in sorted
|
|
59
|
+
order, sequentially. a scenario needs both `task.md` and `criteria.json` to be
|
|
60
|
+
discovered. exit 2 if anything errored, 1 if anything failed, else 0.
|
|
61
|
+
|
|
62
|
+
`EVALS_CONCURRENCY` is passed to promptfoo as `-j` (default 4). it parallelizes
|
|
63
|
+
within one scenario, not across them.
|
|
64
|
+
|
|
65
|
+
one known failure mode: judge calls through a gateway can drop at the transport
|
|
66
|
+
layer ([uinaf/agent-platform#28](https://github.com/uinaf/agent-platform/issues/28)).
|
|
67
|
+
that surfaces as an ERROR with no usable result, not as a graded FAIL, and the
|
|
68
|
+
mitigation is a rerun. `sweep` without `--all` resumes, so a rerun only picks up
|
|
69
|
+
what is missing.
|
|
70
|
+
|
|
71
|
+
## summarize
|
|
72
|
+
|
|
73
|
+
```sh
|
|
74
|
+
skillcheck summarize [--allow-mixed]
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
reduces `<root>/.skillcheck/results/*.json` into
|
|
78
|
+
`<root>/.skillcheck/scorecards/<UTC-date>.json`: one entry per scenario with
|
|
79
|
+
skill, scenario, harness, tree sha, score, pass, both models, latency, tokens.
|
|
80
|
+
|
|
81
|
+
if a scorecard for today already exists, the two are merged on
|
|
82
|
+
`(skill, scenario, harness)`: entries from this run win, entries it did not
|
|
83
|
+
touch survive, and the merge is reported on stdout. summarizing after rerunning
|
|
84
|
+
six of twenty-nine scenarios therefore leaves twenty-nine rows in the file, not
|
|
85
|
+
six. a same-date file that cannot be parsed stops the write instead of being
|
|
86
|
+
overwritten.
|
|
87
|
+
|
|
88
|
+
files that are not promptfoo results are skipped with a warning rather than
|
|
89
|
+
failing the reduction.
|
|
90
|
+
|
|
91
|
+
## provenance
|
|
92
|
+
|
|
93
|
+
each successful run writes a `<name>.meta.json` sidecar next to its result:
|
|
94
|
+
|
|
95
|
+
```json
|
|
96
|
+
{
|
|
97
|
+
"skills_tree_sha": "<root repo HEAD at run time>",
|
|
98
|
+
"harness": "claude",
|
|
99
|
+
"ran_at": "<ISO timestamp>",
|
|
100
|
+
"tool_version": "<skillcheck version>"
|
|
101
|
+
}
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
`summarize` reads those sidecars and refuses to mix skills-tree revisions in one
|
|
105
|
+
scorecard unless `--allow-mixed`, in which case the top-level `skills_tree_sha`
|
|
106
|
+
becomes `mixed` and per-entry shas remain. a result with no sidecar reduces as
|
|
107
|
+
`unattested`.
|
|
108
|
+
|
|
109
|
+
## state
|
|
110
|
+
|
|
111
|
+
`<root>/.skillcheck/` holds `scratch/` and `results/`, both disposable and safe
|
|
112
|
+
to gitignore, and `scorecards/`, which is meant to be committed. nothing is ever
|
|
113
|
+
written inside the installed package.
|
|
114
|
+
|
|
115
|
+
## auth
|
|
116
|
+
|
|
117
|
+
| variable | effect |
|
|
118
|
+
| --------------------------------------------- | ----------------------------------------------------------------- |
|
|
119
|
+
| `ANTHROPIC_BASE_URL` + `ANTHROPIC_AUTH_TOKEN` | claude agent and judge go through a gateway |
|
|
120
|
+
| none of the above | falls back to the local Claude Code session |
|
|
121
|
+
| `ANTHROPIC_API_KEY` | judge grades over `anthropic:messages:<model>` instead of the SDK |
|
|
122
|
+
| `CODEX_HOME` (default `~/.codex`) | where the codex harness finds the local `codex` CLI login |
|
|
123
|
+
| `OPENAI_API_KEY` | codex agent auth when there is no local login |
|
|
124
|
+
|
|
125
|
+
the judge stays on the Anthropic selection regardless of the agent harness.
|
package/package.json
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@uinaf/skillcheck",
|
|
3
|
+
"version": "0.1.3",
|
|
4
|
+
"description": "Lint and eval harness for agent skills",
|
|
5
|
+
"homepage": "https://github.com/uinaf/skillcheck#readme",
|
|
6
|
+
"bugs": {
|
|
7
|
+
"url": "https://github.com/uinaf/skillcheck/issues"
|
|
8
|
+
},
|
|
9
|
+
"license": "MIT",
|
|
10
|
+
"author": "undefined is not a function LLC",
|
|
11
|
+
"repository": {
|
|
12
|
+
"type": "git",
|
|
13
|
+
"url": "git+https://github.com/uinaf/skillcheck.git"
|
|
14
|
+
},
|
|
15
|
+
"bin": {
|
|
16
|
+
"skillcheck": "dist/cli.js"
|
|
17
|
+
},
|
|
18
|
+
"files": [
|
|
19
|
+
"dist",
|
|
20
|
+
"docs",
|
|
21
|
+
"README.md",
|
|
22
|
+
"LICENSE"
|
|
23
|
+
],
|
|
24
|
+
"type": "module",
|
|
25
|
+
"publishConfig": {
|
|
26
|
+
"access": "public",
|
|
27
|
+
"registry": "https://registry.npmjs.org/"
|
|
28
|
+
},
|
|
29
|
+
"scripts": {
|
|
30
|
+
"verify": "vp check && vp pack && vp test run",
|
|
31
|
+
"prepare": "vp config --no-agent",
|
|
32
|
+
"prepublishOnly": "pnpm run verify"
|
|
33
|
+
},
|
|
34
|
+
"dependencies": {
|
|
35
|
+
"@anthropic-ai/claude-agent-sdk": "^0.3.233",
|
|
36
|
+
"@openai/codex-sdk": "^0.147.0",
|
|
37
|
+
"promptfoo": "^0.122.0"
|
|
38
|
+
},
|
|
39
|
+
"devDependencies": {
|
|
40
|
+
"@types/node": "^26.2.0",
|
|
41
|
+
"vite": "catalog:",
|
|
42
|
+
"vite-plus": "catalog:"
|
|
43
|
+
},
|
|
44
|
+
"engines": {
|
|
45
|
+
"node": ">=24"
|
|
46
|
+
},
|
|
47
|
+
"packageManager": "pnpm@11.22.0"
|
|
48
|
+
}
|