@uinaf/skillcheck 0.4.0 → 0.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -2,25 +2,28 @@
2
2
 
3
3
  # uinaf/skillcheck
4
4
 
5
- lint and eval harness for agent skills. one CLI with two halves: a keyless
5
+ Lint and eval harness for agent skills. One CLI with two halves: a keyless
6
6
  structural lint any repo can run in CI, and a promptfoo-driven eval loop that
7
7
  grades what a skill actually makes an agent do.
8
8
 
9
- built for [uinaf](https://uinaf.dev) skill repos. nothing in it is
10
- uinaf-specific. it ships no opinion about what a skill should say, only about
9
+ Built for [uinaf](https://uinaf.dev) skill repos. Nothing in it is
10
+ uinaf-specific. It ships no opinion about what a skill should say, only about
11
11
  where skills sit and how a scenario is scored.
12
12
 
13
- ## install
13
+ ## Install
14
14
 
15
15
  ```sh
16
16
  pnpm add -D @uinaf/skillcheck
17
17
  ```
18
18
 
19
- node 24 or newer. the package ships compiled ESM and runs no install scripts,
20
- so a runner using `--ignore-scripts` is fine. consumers still pinned to the
21
- pre-npm git tags are covered in [adoption](docs/adoption.md).
19
+ Node 24 or newer. The package ships compiled ESM, runs no install scripts, and
20
+ has no regular dependencies, so a runner using `--ignore-scripts` is fine and
21
+ the lint-only install stays at a handful of packages. The promptfoo eval engine
22
+ and provider SDKs are optional peers, installed only on the operator machine
23
+ that runs evals ([adoption](docs/adoption.md)). Consumers still pinned to the
24
+ pre-npm git tags are covered there too.
22
25
 
23
- ## use
26
+ ## Use
24
27
 
25
28
  ```sh
26
29
  skillcheck lint # structural lint, no credentials
@@ -29,12 +32,12 @@ skillcheck sweep && skillcheck summarize # every scenario, then a scorecard
29
32
  ```
30
33
 
31
34
  `lint` needs nothing. `run` and `sweep` need model auth, which is why sweeps
32
- stay operator-run and consumer repos never hold credentials. every command,
35
+ stay operator-run and consumer repos never hold credentials. Every command,
33
36
  flag, and auth variable is in [usage](docs/usage.md).
34
37
 
35
- ## layout contract
38
+ ## Layout contract
36
39
 
37
- frozen, not configurable. every command reads one root: `--root <dir>`, or the
40
+ Frozen, not configurable. Every command reads one root: `--root <dir>`, or the
38
41
  current directory.
39
42
 
40
43
  ```text
@@ -47,18 +50,18 @@ current directory.
47
50
  `cli/*/skills/<skill>/` is scanned too, for repos that keep a skill next to the
48
51
  CLI it documents.
49
52
 
50
- ## docs
53
+ ## Docs
51
54
 
52
- | doc | when |
55
+ | Doc | When |
53
56
  | --------------------------------------------------------------- | ------------------------------------- |
54
- | [usage](docs/usage.md) | every subcommand, flag, and auth path |
55
- | [scenarios](docs/scenarios.md) | writing an eval scenario |
56
- | [authoring](docs/authoring.md) | writing and auditing the skill itself |
57
- | [adoption](docs/adoption.md) | wiring the lint into another repo |
58
- | [releasing](docs/releasing.md) | the npm pipeline |
59
- | [contributing](CONTRIBUTING.md) | local setup and the verify gate |
60
- | [security](https://github.com/uinaf/skillcheck/security/policy) | reporting a vulnerability |
61
-
62
- ## license
57
+ | [Usage](docs/usage.md) | Every subcommand, flag, and auth path |
58
+ | [Scenarios](docs/scenarios.md) | Writing an eval scenario |
59
+ | [Authoring](docs/authoring.md) | Writing and auditing the skill itself |
60
+ | [Adoption](docs/adoption.md) | Wiring the lint into another repo |
61
+ | [Releasing](docs/releasing.md) | The npm pipeline |
62
+ | [Contributing](CONTRIBUTING.md) | Local setup and the verify gate |
63
+ | [Security](https://github.com/uinaf/skillcheck/security/policy) | Reporting a vulnerability |
64
+
65
+ ## License
63
66
 
64
67
  MIT · undefined is not a function LLC
package/dist/cli.js CHANGED
@@ -1,6 +1,6 @@
1
1
  #!/usr/bin/env node
2
2
  import { lintSkills } from "./lint.js";
3
- import { generateRun, runNameFor } from "./scenario.js";
3
+ import { generateRun, requiredEvalPackages, resolvePackageDir, runNameFor } from "./scenario.js";
4
4
  import { execFileSync, spawnSync } from "node:child_process";
5
5
  import fs from "node:fs";
6
6
  import path from "node:path";
@@ -81,12 +81,35 @@ function runOptions(flags) {
81
81
  maxTurns: flags.has("--max-turns") ? parseMaxTurns(flags.get("--max-turns")) : void 0
82
82
  };
83
83
  }
84
+ function ensureEvalPackages(opts) {
85
+ const missing = requiredEvalPackages(opts, process.env.ANTHROPIC_API_KEY !== void 0).filter((pkg) => resolvePackageDir(pkg) === void 0);
86
+ if (missing.length === 0) return;
87
+ const peers = JSON.parse(fs.readFileSync(path.join(packageDir, "package.json"), "utf8")).peerDependencies ?? {};
88
+ const specs = missing.map((pkg) => `"${pkg}@${peers[pkg] ?? "latest"}"`).join(" ");
89
+ console.error([
90
+ `missing eval package(s): ${missing.join(", ")}`,
91
+ "",
92
+ "The eval engine is an optional peer so `skillcheck lint` installs stay",
93
+ "small. Evals are operator-run; install the peers next to @uinaf/skillcheck:",
94
+ "",
95
+ ` pnpm add -D ${specs}`
96
+ ].join("\n"));
97
+ process.exit(2);
98
+ }
99
+ function promptfooEntry() {
100
+ const dir = resolvePackageDir("promptfoo");
101
+ if (dir === void 0) throw new Error("promptfoo is not installed");
102
+ const bin = JSON.parse(fs.readFileSync(path.join(dir, "package.json"), "utf8")).bin;
103
+ const rel = typeof bin === "string" ? bin : bin?.promptfoo;
104
+ if (typeof rel !== "string") throw new Error(`promptfoo at ${dir} declares no bin`);
105
+ return path.join(dir, rel);
106
+ }
84
107
  function classifyResult(raw) {
85
108
  const root = raw;
86
109
  const res = root?.results?.results?.[0];
87
- if (res === void 0) return { error: "promptfoo output carried no result" };
110
+ if (res === void 0 || res === null) return { error: "promptfoo output carried no result" };
88
111
  const message = typeof res.error === "string" ? res.error.trim() : "";
89
- const stats = root?.results?.stats;
112
+ const stats = root?.results?.stats ?? void 0;
90
113
  const gradedByStats = stats !== void 0 && (stats.errors ?? 0) === 0 && ((stats.failures ?? 0) > 0 || (stats.successes ?? 0) > 0);
91
114
  if (message !== "" && !gradedByStats) return { error: message };
92
115
  if (stats !== void 0 && (stats.errors ?? 0) > 0 && (stats.successes ?? 0) === 0 && (stats.failures ?? 0) === 0) return { error: "promptfoo reported an errored test with nothing graded" };
@@ -105,6 +128,9 @@ function gitHead(root) {
105
128
  function metaPath(resultPath) {
106
129
  return resultPath.replace(/\.json$/, ".meta.json");
107
130
  }
131
+ function attemptPath(resultPath) {
132
+ return `${resultPath}.attempt`;
133
+ }
108
134
  function runScenario(scenarioDir, opts, root) {
109
135
  const dirs = stateDirs(root);
110
136
  const { name, configPath } = generateRun(path.resolve(scenarioDir), opts, {
@@ -114,11 +140,12 @@ function runScenario(scenarioDir, opts, root) {
114
140
  });
115
141
  fs.mkdirSync(dirs.results, { recursive: true });
116
142
  const resultPath = path.join(dirs.results, `${name}.json`);
143
+ fs.writeFileSync(attemptPath(resultPath), "{}\n");
117
144
  fs.rmSync(resultPath, { force: true });
118
145
  fs.rmSync(metaPath(resultPath), { force: true });
119
146
  const sha = gitHead(root);
120
- const rc = spawnSync("npx", [
121
- "promptfoo",
147
+ const rc = spawnSync(process.execPath, [
148
+ promptfooEntry(),
122
149
  "eval",
123
150
  "--no-cache",
124
151
  "--no-progress-bar",
@@ -151,12 +178,15 @@ function runScenario(scenarioDir, opts, root) {
151
178
  outcome.score = verdict.score;
152
179
  outcome.pass = verdict.pass;
153
180
  outcome.error = verdict.error;
154
- if (verdict.score !== void 0) fs.writeFileSync(metaPath(resultPath), JSON.stringify({
155
- skills_tree_sha: sha,
156
- harness: opts.harness,
157
- ran_at: (/* @__PURE__ */ new Date()).toISOString(),
158
- tool_version: toolVersion()
159
- }, null, 2) + "\n");
181
+ if (verdict.score !== void 0) {
182
+ fs.writeFileSync(metaPath(resultPath), JSON.stringify({
183
+ skills_tree_sha: sha,
184
+ harness: opts.harness,
185
+ ran_at: (/* @__PURE__ */ new Date()).toISOString(),
186
+ tool_version: toolVersion()
187
+ }, null, 2) + "\n");
188
+ fs.rmSync(attemptPath(resultPath), { force: true });
189
+ }
160
190
  return outcome;
161
191
  }
162
192
  function discoverScenarios(root) {
@@ -182,7 +212,9 @@ function discoverScenarios(root) {
182
212
  function cmdRun(argv) {
183
213
  const { positional, flags } = parseArgs(argv);
184
214
  if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--judge-effort EFFORT] [--harness claude|codex|cursor]");
185
- const o = runScenario(positional[0], runOptions(flags), resolveRoot(flags));
215
+ const opts = runOptions(flags);
216
+ ensureEvalPackages(opts);
217
+ const o = runScenario(positional[0], opts, resolveRoot(flags));
186
218
  if (o.score === void 0) {
187
219
  console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
188
220
  process.exit(2);
@@ -195,6 +227,7 @@ function cmdSweep(argv) {
195
227
  if (positional.length > 0) fail("usage: skillcheck sweep [--root DIR] [--all]");
196
228
  const root = resolveRoot(flags);
197
229
  const opts = runOptions(flags);
230
+ ensureEvalPackages(opts);
198
231
  const all = flags.get("--all") === true;
199
232
  const resultsDir = stateDirs(root).results;
200
233
  let passed = 0, failed = 0, errored = 0, skipped = 0;
@@ -221,11 +254,30 @@ function cmdSweep(argv) {
221
254
  console.log(`\nsweep: ${passed} passed, ${failed} failed, ${errored} errored, ${skipped} skipped`);
222
255
  process.exit(errored > 0 ? 2 : failed > 0 ? 1 : 0);
223
256
  }
257
+ function resultIdentity(file) {
258
+ const base = file.replace(/\.json$/, "");
259
+ const suffix = base.match(/--(codex|cursor)$/);
260
+ const harness = suffix === null ? "claude" : suffix[1];
261
+ const [skill, ...rest] = base.replace(/--(codex|cursor)$/, "").split("--");
262
+ return {
263
+ skill,
264
+ scenario: rest.join("--"),
265
+ harness
266
+ };
267
+ }
224
268
  function reduceResults(dir, allowMixed) {
225
269
  const entries = [];
226
270
  const skipped = [];
227
271
  const shas = /* @__PURE__ */ new Set();
228
- for (const f of fs.readdirSync(dir).filter((f) => f.endsWith(".json") && !f.endsWith(".meta.json")).sort()) {
272
+ const files = fs.readdirSync(dir);
273
+ const incomplete = new Set(files.filter((f) => f.endsWith(".json.attempt")).map((f) => f.replace(/\.attempt$/, "")));
274
+ const results = files.filter((f) => f.endsWith(".json") && !f.endsWith(".meta.json"));
275
+ for (const f of [.../* @__PURE__ */ new Set([...results, ...incomplete])].sort()) {
276
+ if (incomplete.has(f)) {
277
+ console.error(`skipping ${f}: attempt did not complete with a graded result`);
278
+ skipped.push(f);
279
+ continue;
280
+ }
229
281
  let raw;
230
282
  try {
231
283
  raw = JSON.parse(fs.readFileSync(path.join(dir, f), "utf8"));
@@ -233,17 +285,16 @@ function reduceResults(dir, allowMixed) {
233
285
  raw = void 0;
234
286
  }
235
287
  const res = raw?.results?.results?.[0];
236
- if (typeof res?.score !== "number" || typeof res?.success !== "boolean") {
237
- console.error(`skipping ${f}: not a promptfoo result`);
288
+ const verdict = classifyResult(raw);
289
+ if (verdict.score === void 0 || verdict.pass === void 0) {
290
+ console.error(`skipping ${f}: ${verdict.error}`);
238
291
  skipped.push(f);
239
292
  continue;
240
293
  }
241
294
  const provider = raw.config?.providers?.[0];
242
295
  const judge = raw.config?.defaultTest?.options?.provider;
243
296
  const base = f.replace(/\.json$/, "");
244
- const suffix = base.match(/--(codex|cursor)$/);
245
- const harness = suffix === null ? "claude" : suffix[1];
246
- const [skill, ...rest] = base.replace(/--(codex|cursor)$/, "").split("--");
297
+ const { skill, scenario, harness } = resultIdentity(f);
247
298
  let sha = "unattested";
248
299
  try {
249
300
  sha = JSON.parse(fs.readFileSync(path.join(dir, `${base}.meta.json`), "utf8")).skills_tree_sha ?? "unattested";
@@ -251,11 +302,11 @@ function reduceResults(dir, allowMixed) {
251
302
  shas.add(sha);
252
303
  entries.push({
253
304
  skill,
254
- scenario: rest.join("--"),
305
+ scenario,
255
306
  harness,
256
307
  skills_tree_sha: sha,
257
- score: res.score,
258
- pass: res.success,
308
+ score: verdict.score,
309
+ pass: verdict.pass,
259
310
  agent_model: provider?.config?.model ?? `${harness}-default`,
260
311
  judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? judge?.id ?? "unknown",
261
312
  latency_ms: res.latencyMs,
@@ -309,15 +360,19 @@ function cmdSummarize(argv) {
309
360
  const { positional, flags } = parseArgs(argv);
310
361
  if (positional.length > 0) fail("usage: skillcheck summarize [--root DIR] [--allow-mixed]");
311
362
  const dirs = stateDirs(resolveRoot(flags));
312
- if (!fs.existsSync(dirs.results)) fail(`no results directory at ${dirs.results} — run some evals first`);
363
+ if (!fs.existsSync(dirs.results)) fail(`no results directory at ${dirs.results}; run some evals first`);
313
364
  const { entries, skipped } = reduceResults(dirs.results, flags.get("--allow-mixed") === true);
314
365
  fs.mkdirSync(dirs.scorecards, { recursive: true });
315
366
  const out = path.join(dirs.scorecards, `${(/* @__PURE__ */ new Date()).toISOString().slice(0, 10)}.json`);
316
367
  const existing = readExistingScorecard(out);
368
+ const skippedKeys = new Set(skipped.map((file) => entryKey(resultIdentity(file))));
369
+ if (existing.some((entry) => skippedKeys.has(entryKey(entry)))) throw new Error("skipped rerun matches an existing score; refusing to carry it or overwrite the scorecard");
317
370
  const merged = mergeScorecard(existing, entries);
371
+ const treeSha = treeShaOf(merged.entries);
372
+ if (treeSha === "mixed" && flags.get("--allow-mixed") !== true) throw new Error("scorecard spans multiple skills-tree revisions; rerun stale ones or pass --allow-mixed");
318
373
  const scorecard = {
319
374
  ran_at: (/* @__PURE__ */ new Date()).toISOString(),
320
- skills_tree_sha: treeShaOf(merged.entries),
375
+ skills_tree_sha: treeSha,
321
376
  scenarios: merged.entries
322
377
  };
323
378
  fs.writeFileSync(out, JSON.stringify(scorecard, null, 2) + "\n");
package/dist/scenario.js CHANGED
@@ -213,8 +213,25 @@ function buildConfig(s, workdir, manifestPath, opts, paths) {
213
213
  }
214
214
  const SDK_PACKAGES = ["@anthropic-ai/claude-agent-sdk", "@openai/codex-sdk"];
215
215
  function sdkNodeModulesDir() {
216
+ for (const pkg of SDK_PACKAGES) {
217
+ const dir = holdingNodeModules(pkg);
218
+ if (dir !== void 0) return dir;
219
+ }
220
+ }
221
+ function holdingNodeModules(pkg) {
216
222
  const require = createRequire(import.meta.url);
217
- for (const pkg of SDK_PACKAGES) for (const dir of require.resolve.paths(pkg) ?? []) if (fs.existsSync(path.join(dir, pkg, "package.json"))) return dir;
223
+ for (const dir of require.resolve.paths(pkg) ?? []) if (fs.existsSync(path.join(dir, pkg, "package.json"))) return dir;
224
+ }
225
+ function resolvePackageDir(pkg) {
226
+ const dir = holdingNodeModules(pkg);
227
+ return dir === void 0 ? void 0 : path.join(dir, pkg);
228
+ }
229
+ function requiredEvalPackages(opts, hasAnthropicKey) {
230
+ const pkgs = ["promptfoo"];
231
+ const sdkJudge = !opts.judgeModel.includes(":") && !hasAnthropicKey;
232
+ if (opts.harness === "claude" || sdkJudge) pkgs.push("@anthropic-ai/claude-agent-sdk");
233
+ if (opts.harness === "codex") pkgs.push("@openai/codex-sdk");
234
+ return pkgs;
218
235
  }
219
236
  function generateRun(scenarioDir, opts, paths) {
220
237
  const s = loadScenario(scenarioDir);
@@ -239,4 +256,4 @@ function generateRun(scenarioDir, opts, paths) {
239
256
  };
240
257
  }
241
258
  //#endregion
242
- export { buildConfig, generateRun, isHiddenSkill, loadScenario, materialize, runNameFor, sdkNodeModulesDir, stripHiddenFlag };
259
+ export { buildConfig, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag };
package/docs/adoption.md CHANGED
@@ -1,15 +1,15 @@
1
- # adopting it in a repo
1
+ # Adopting it in a repo
2
2
 
3
- two separate decisions: run the lint in CI, and run evals on a machine that has
4
- model auth. only the first belongs in a consumer repo.
3
+ Two separate decisions: run the lint in CI, and run evals on a machine that has
4
+ model auth. Only the first belongs in a consumer repo.
5
5
 
6
- ## lint in CI
6
+ ## Lint in CI
7
7
 
8
8
  ```sh
9
9
  pnpm add -D @uinaf/skillcheck
10
10
  ```
11
11
 
12
- runners that install with `--ignore-scripts` are fine: the package ships
12
+ Runners that install with `--ignore-scripts` are fine: the package ships
13
13
  compiled ESM and has no install, prepare, or postinstall script.
14
14
 
15
15
  ```json
@@ -32,49 +32,59 @@ jobs:
32
32
  - run: pnpm run skills:lint
33
33
  ```
34
34
 
35
- the job needs no secrets and no network beyond the install. node 24 is the
35
+ The job needs no secrets and no network beyond the install. Node 24 is the
36
36
  floor.
37
37
 
38
- run it through the script rather than a bare `npx skillcheck`: the script
38
+ Run it through the script rather than a bare `npx skillcheck`: the script
39
39
  resolves the version the repo pinned, and `npx` would resolve the latest one on
40
40
  the registry.
41
41
 
42
- ## evals
42
+ ## Evals
43
43
 
44
- sweeps need model credentials, so they stay off consumer CI and run from an
45
- operator machine or a job that already holds gateway auth:
44
+ Sweeps need model credentials, so they stay off consumer CI and run from an
45
+ operator machine or a job that already holds gateway auth. They also need the
46
+ eval engine, which is an optional peer precisely so the lint-only install
47
+ above stays small. Install it next to the package on the operator machine:
48
+
49
+ ```sh
50
+ pnpm add -D promptfoo @anthropic-ai/claude-agent-sdk @openai/codex-sdk
51
+ ```
52
+
53
+ `run` and `sweep` check for the peers the selected harness needs and exit 2
54
+ with that install command when they are missing, so a lint-only install never
55
+ crashes into a resolution error. Then:
46
56
 
47
57
  ```sh
48
58
  skillcheck sweep # resumes: only scenarios without results
49
59
  skillcheck summarize # writes .skillcheck/scorecards/<UTC-date>.json
50
60
  ```
51
61
 
52
- commit `.skillcheck/scorecards/`. gitignore the rest:
62
+ Commit `.skillcheck/scorecards/`. Gitignore the rest:
53
63
 
54
64
  ```gitignore
55
65
  .skillcheck/results/
56
66
  .skillcheck/scratch/
57
67
  ```
58
68
 
59
- a scorecard is only comparable against the tree it graded, which is why every
69
+ A scorecard is only comparable against the tree it graded, which is why every
60
70
  result carries the root repo's HEAD and `summarize` refuses to mix revisions
61
71
  without `--allow-mixed`.
62
72
 
63
- ## upgrading
73
+ ## Upgrading
64
74
 
65
- bump the version in `package.json` and rerun the sweep. results carry the
75
+ Bump the version in `package.json` and rerun the sweep. Results carry the
66
76
  `tool_version` that produced them, so a scorecard says which harness build it
67
77
  came from as well as which skills tree.
68
78
 
69
- ## the older install path
79
+ ## The older install path
70
80
 
71
- before the package was published, consumers installed it from a git tag:
81
+ Before the package was published, consumers installed it from a git tag:
72
82
 
73
83
  ```sh
74
84
  npm i -D github:uinaf/skillcheck#v0.1.3
75
85
  ```
76
86
 
77
- those tags are frozen and still work: they carry a committed `dist/`, and npm
78
- 12 consumers needed `allow-git=root` in `.npmrc` to accept the spec at all. the
79
- registry install needs none of that. tags from `v0.1.4` on are npm releases and
87
+ Those tags are frozen and still work: they carry a committed `dist/`, and npm
88
+ 12 consumers needed `allow-git=root` in `.npmrc` to accept the spec at all. The
89
+ registry install needs none of that. Tags from `v0.1.4` on are npm releases and
80
90
  carry no `dist/`, so a git spec pointing at one will not run.
package/docs/authoring.md CHANGED
@@ -1,70 +1,70 @@
1
- # authoring and auditing skills
1
+ # Authoring and auditing skills
2
2
 
3
- what `lint` and evals cannot judge: whether a skill is worth routing to and
4
- cheap to load. use this when writing a skill or auditing one. evidence beats
3
+ What `lint` and evals cannot judge: whether a skill is worth routing to and
4
+ cheap to load. Use this when writing a skill or auditing one. Evidence beats
5
5
  stylistic preference; run `skillcheck lint` first and let this cover the rest.
6
6
 
7
- ## metadata and discovery
7
+ ## Metadata and discovery
8
8
 
9
9
  - `name` is concrete and easy to say out loud. `helper`, `tools`, `utils` are
10
10
  discovery smells.
11
11
  - `description` is third person and says both what the skill does and when to
12
- use it. it is an always-loaded retrieval pointer: front-load the concrete
12
+ use it. It is an always-loaded retrieval pointer: front-load the concrete
13
13
  action or domain that should activate it.
14
- - one trigger per materially distinct request branch. collapse synonyms that
14
+ - One trigger per materially distinct request branch. Collapse synonyms that
15
15
  rename the same branch.
16
- - state the main overlap boundary without naming another skill.
16
+ - State the main overlap boundary without naming another skill.
17
17
 
18
- ## body shape
18
+ ## Body shape
19
19
 
20
- - keep `SKILL.md` on workflow, principles, boundaries, and routing. lead with
20
+ - Keep `SKILL.md` on workflow, principles, boundaries, and routing. Lead with
21
21
  the task, not a bibliography.
22
- - assume the model is smart; spend tokens on repo-specific judgment. delete any
22
+ - Assume the model is smart; spend tokens on repo-specific judgment. Delete any
23
23
  instruction that would not change a capable model's behavior.
24
- - match freedom to risk: high for contextual judgment, medium when a preferred
24
+ - Match freedom to risk: high for contextual judgment, medium when a preferred
25
25
  pattern exists, low for fragile operations.
26
- - say what evidence to gather and what a complete result includes. end each
26
+ - Say what evidence to gather and what a complete result includes. End each
27
27
  step with an observable completion condition, not "understood" or "handled".
28
28
 
29
- ## progressive disclosure
29
+ ## Progressive disclosure
30
30
 
31
- - durable detail, rubrics, and long examples go in `references/`, one hop from
32
- `SKILL.md`, each with a task-shaped retrieval job. material every path needs
31
+ - Durable detail, rubrics, and long examples go in `references/`, one hop from
32
+ `SKILL.md`, each with a task-shaped retrieval job. Material every path needs
33
33
  stays inline.
34
- - for repeated deterministic work, route to the target's existing framework,
34
+ - For repeated deterministic work, route to the target's existing framework,
35
35
  schema, task graph, or library; otherwise add a tested module in the
36
36
  project's primary language, not ad-hoc shell rendered as prose.
37
- - when executable code belongs to another maintained project, link the exact
37
+ - When executable code belongs to another maintained project, link the exact
38
38
  public artifact and state the contract it demonstrates; do not fork it into
39
39
  prose.
40
- - a package stays independently usable: state prerequisites and out-of-scope
41
- next steps locally. never invoke, import, or assume a sibling skill.
40
+ - A package stays independently usable: state prerequisites and out-of-scope
41
+ next steps locally. Never invoke, import, or assume a sibling skill.
42
42
 
43
- ## audit
43
+ ## Audit
44
44
 
45
- grade each dimension strong, mixed, or weak:
45
+ Grade each dimension strong, mixed, or weak:
46
46
 
47
- | dimension | question |
47
+ | Dimension | Question |
48
48
  | ---------------------- | --------------------------------------------------------------------- |
49
- | discovery | does metadata alone route a realistic request here |
50
- | workflow | does the body say how to begin, what evidence to gather, when to stop |
51
- | progressive disclosure | is detail in the right file |
52
- | repo fit | are links, commands, and conventions current |
53
- | verification | is the strongest mechanical check named, plus a real evidence loop |
54
- | boundaries | are limits and next steps stated without leaning on a sibling skill |
49
+ | Discovery | Does metadata alone route a realistic request here |
50
+ | Workflow | Does the body say how to begin, what evidence to gather, when to stop |
51
+ | Progressive disclosure | Is detail in the right file |
52
+ | Repo fit | Are links, commands, and conventions current |
53
+ | Verification | Is the strongest mechanical check named, plus a real evidence loop |
54
+ | Boundaries | Are limits and next steps stated without leaning on a sibling skill |
55
55
 
56
- blockers, must-fix: invalid frontmatter; a description that fails discovery;
56
+ Blockers, must-fix: invalid frontmatter; a description that fails discovery;
57
57
  stale commands, paths, or links; a workflow with no start, evidence loop, or
58
58
  completion; conflicts with the repo's guidance; sibling-skill dependencies.
59
59
 
60
- major findings: vague name; synonym-stuffed description; bloated `SKILL.md`;
60
+ Major findings: vague name; synonym-stuffed description; bloated `SKILL.md`;
61
61
  missing or muddy boundaries; prose re-inventing a deterministic tool; abstract
62
62
  examples.
63
63
 
64
- ## improve
64
+ ## Improve
65
65
 
66
- fix blockers first, then the highest-leverage majors. prefer the smallest
67
- change that improves activation, decision quality, or proof. when pruning,
66
+ Fix blockers first, then the highest-leverage majors. Prefer the smallest
67
+ change that improves activation, decision quality, or proof. When pruning,
68
68
  measure common-path context for representative requests; line count alone does
69
- not reveal retrieval cost. after edits, rerun `skillcheck lint` and the repo's
69
+ not reveal retrieval cost. After edits, rerun `skillcheck lint` and the repo's
70
70
  gate, and rerun evals when behavior was the thing changed.
package/docs/releasing.md CHANGED
@@ -1,8 +1,8 @@
1
- # releasing
1
+ # Releasing
2
2
 
3
- ## pipeline
3
+ ## Pipeline
4
4
 
5
- a push to `main` runs one workflow, `.github/workflows/release.yml`:
5
+ A push to `main` runs one workflow, `.github/workflows/release.yml`:
6
6
 
7
7
  ```text
8
8
  verify ──┐
@@ -12,69 +12,69 @@ scan ────┘
12
12
 
13
13
  `verify` and `scan` are the shared gate: `verify` is called from `verify.yml`,
14
14
  and `scan` calls the shared scan in `uinaf/.github`, the same one `scan.yml`
15
- runs for pull requests. keep it that way: a second copy of the gate on a push-to-`main` workflow
15
+ runs for pull requests. Keep it that way: a second copy of the gate on a push-to-`main` workflow
16
16
  races this one over the same commit.
17
17
 
18
- the file name `release.yml` is load-bearing. see below.
18
+ The file name `release.yml` is load-bearing. See below.
19
19
 
20
20
  ## npm
21
21
 
22
22
  `@uinaf/skillcheck` publishes from `.github/workflows/release.yml` via npm
23
- Trusted Publishing (OIDC) and the `uinaf-releaser` GitHub App. there is no npm
23
+ Trusted Publishing (OIDC) and the `uinaf-releaser` GitHub App. There is no npm
24
24
  token in this repository, in its environments, or in the organization.
25
25
 
26
- required on the `release` GitHub Environment:
26
+ Required on the `release` GitHub Environment:
27
27
 
28
- | name | kind | purpose |
28
+ | Name | Kind | Purpose |
29
29
  | ------------------------------- | ------ | ------------------------------------------- |
30
30
  | `UINAF_RELEASE_APP_CLIENT_ID` | var | GitHub App client id for the releaser bot |
31
31
  | `UINAF_RELEASE_APP_PRIVATE_KEY` | secret | GitHub App private key for the releaser bot |
32
32
 
33
- the trusted publisher on npmjs.com is registered by **file path**, so
33
+ The trusted publisher on npmjs.com is registered by **file path**, so
34
34
  `.github/workflows/release.yml` cannot be renamed or moved without editing that
35
- registration first. a rename fails the publish with an identity mismatch, and
36
- nothing earlier in the run reports it. the `release` environment name is bound
35
+ registration first. A rename fails the publish with an identity mismatch, and
36
+ nothing earlier in the run reports it. The `release` environment name is bound
37
37
  the same way.
38
38
 
39
- deleting the `release` environment deletes both rows above with it, and there is
39
+ Deleting the `release` environment deletes both rows above with it, and there is
40
40
  no repo-level fallback: `create-github-app-token` then runs with empty inputs
41
- and the job fails at that step. the private key cannot be read back from
41
+ and the job fails at that step. The private key cannot be read back from
42
42
  GitHub; recreating it means generating a new one in the App settings.
43
43
 
44
- ## version history
44
+ ## Version history
45
45
 
46
- semantic-release owns the version and the tag. `tagFormat` is `v${version}` and
46
+ The version and the tag are owned by semantic-release. `tagFormat` is `v${version}` and
47
47
  history continues from `v0.1.3`; `v0.1.0`–`v0.1.3` are the legacy git-install
48
48
  tags and are never deleted or moved.
49
49
 
50
- during preparation, `@semantic-release/npm` stages the released `package.json`
50
+ During preparation, `@semantic-release/npm` stages the released `package.json`
51
51
  version and `@jno21/semantic-release-github-commit` commits it to `main` through
52
52
  GitHub's API as the authenticated App. GitHub signs that commit, and the release
53
- tag points to it. the `[skip ci]` marker on that commit is what stops a release
53
+ tag points to it. The `[skip ci]` marker on that commit is what stops a release
54
54
  from releasing itself.
55
55
 
56
- check what the next version would be without publishing anything:
56
+ Check what the next version would be without publishing anything:
57
57
 
58
58
  ```sh
59
59
  pnpm dlx semantic-release --dry-run --no-ci
60
60
  ```
61
61
 
62
- ## the artifact
62
+ ## The artifact
63
63
 
64
64
  `dist/` is generated and untracked. `prepublishOnly` runs `pnpm run verify`,
65
65
  which builds it, so the tarball is always packed from a tree that just passed
66
66
  the gate. `files` is `dist`, `docs`, `README.md`, `LICENSE`.
67
67
 
68
- manual publish is emergency recovery only:
68
+ Manual publish is emergency recovery only:
69
69
 
70
70
  ```sh
71
71
  pnpm run verify
72
72
  npm publish --access public
73
73
  ```
74
74
 
75
- ## the old install path
75
+ ## The old install path
76
76
 
77
- before `@uinaf/skillcheck` existed, consumers installed
78
- `github:uinaf/skillcheck#v0.1.3`. those tags still resolve and still carry a
79
- committed `dist/`, so anything pinned to them keeps working untouched. new
77
+ Before `@uinaf/skillcheck` existed, consumers installed
78
+ `github:uinaf/skillcheck#v0.1.3`. Those tags still resolve and still carry a
79
+ committed `dist/`, so anything pinned to them keeps working untouched. New
80
80
  consumers use npm.
package/docs/scenarios.md CHANGED
@@ -1,20 +1,20 @@
1
- # writing scenarios
1
+ # Writing scenarios
2
2
 
3
- a scenario is two files in a frozen location:
3
+ A scenario is two files in a frozen location:
4
4
 
5
5
  ```text
6
6
  <root>/skills/<skill>/evals/<scenario>/task.md
7
7
  <root>/skills/<skill>/evals/<scenario>/criteria.json
8
8
  ```
9
9
 
10
- the path is the identity: `<skill>--<scenario>` names the run, the result file,
11
- and the scorecard entry. on the codex and cursor harnesses the name gains a
10
+ The path is the identity: `<skill>--<scenario>` names the run, the result file,
11
+ and the scorecard entry. On the codex and cursor harnesses the name gains a
12
12
  `--codex` or `--cursor` suffix, so every harness can hold results side by side.
13
- a directory missing either file is not discovered.
13
+ A directory missing either file is not discovered.
14
14
 
15
15
  ## task.md
16
16
 
17
- the prompt handed to the agent, verbatim, with one piece of syntax. input files
17
+ The prompt handed to the agent, verbatim, with one piece of syntax. Input files
18
18
  are embedded inline and materialized into the workdir before the run:
19
19
 
20
20
  ```md
@@ -25,13 +25,13 @@ Fix the failing check in the config below.
25
25
  ======= END FILE =======
26
26
  ```
27
27
 
28
- each block is replaced in the prompt with a pointer ("Input file `config.json`
29
- is available in your working directory.") and written to disk. destinations
28
+ Each block is replaced in the prompt with a pointer ("Input file `config.json`
29
+ is available in your working directory.") and written to disk. Destinations
30
30
  must stay under the workdir, must not collide, and must not target `.claude/`,
31
31
  `.agents/`, or `.cursor/`, since a fixture that writes agent config would be
32
32
  configuring its own examiner.
33
33
 
34
- write the task the way a user would write it. do not name the skill, describe
34
+ Write the task the way a user would write it. Do not name the skill, describe
35
35
  its steps, or hint at the checklist: routing is part of what is being measured.
36
36
 
37
37
  ## criteria.json
@@ -46,45 +46,45 @@ its steps, or hint at the checklist: routing is part of what is being measured.
46
46
  }
47
47
  ```
48
48
 
49
- `type` must be `weighted_checklist` and the checklist must be non-empty. every
49
+ `type` must be `weighted_checklist` and the checklist must be non-empty. Every
50
50
  item needs a non-empty `name` and `description` and a positive `max_score`.
51
51
 
52
- each item becomes one `llm-rubric` assertion weighted by `max_score`, inside an
53
- assert-set with threshold 0.7. a separate `skill-used` assertion sits outside
52
+ Each item becomes one `llm-rubric` assertion weighted by `max_score`, inside an
53
+ assert-set with threshold 0.7. A separate `skill-used` assertion sits outside
54
54
  that aggregate, so a run that produces good output without ever loading the
55
- skill still fails. there is no test-level threshold: both must pass.
55
+ skill still fails. There is no test-level threshold: both must pass.
56
56
 
57
- write descriptions a judge can check against the deliverable: an observable
58
- property, not a feeling. weight the items that would make a reviewer reject the
57
+ Write descriptions a judge can check against the deliverable: an observable
58
+ property, not a feeling. Weight the items that would make a reviewer reject the
59
59
  work.
60
60
 
61
- ## what the judge sees
61
+ ## What the judge sees
62
62
 
63
- the agent's final message, plus every file in the workdir that differs from the
64
- pre-run manifest. unchanged inputs are omitted; deleted inputs, unreadable
63
+ The agent's final message, plus every file in the workdir that differs from the
64
+ pre-run manifest. Unchanged inputs are omitted; deleted inputs, unreadable
65
65
  files, and non-regular files are named rather than read.
66
66
 
67
- sections are sorted by path, each file is capped at 4,000 characters and the
68
- appended total at 24,000, with truncation stated inline. very large outputs make
69
- rubric judges return nothing at all, which is why the caps exist. keep fixtures
67
+ Sections are sorted by path, each file is capped at 4,000 characters and the
68
+ appended total at 24,000, with truncation stated inline. Very large outputs make
69
+ rubric judges return nothing at all, which is why the caps exist. Keep fixtures
70
70
  small enough that the deliverable fits.
71
71
 
72
- ## hidden skills
72
+ ## Hidden skills
73
73
 
74
- a skill with `disable-model-invocation: true` is explicit-invoke-only in
75
- production, which the agent SDK cannot simulate. so the eval copy, never the shipped one,
74
+ A skill with `disable-model-invocation: true` is explicit-invoke-only in
75
+ production, which the agent SDK cannot simulate. So the eval copy, never the shipped one,
76
76
  has the flag stripped, and the task gains a leading
77
- `Use the <skill> skill for this task.` the eval then measures
78
- behavior-when-invoked rather than routing. the flag is only honored inside the
77
+ `Use the <skill> skill for this task.` The eval then measures
78
+ behavior-when-invoked rather than routing. The flag is only honored inside the
79
79
  frontmatter block; body text mentioning the key does not count.
80
80
 
81
- ## the workdir
81
+ ## The workdir
82
82
 
83
- per run, under `<root>/.skillcheck/scratch/<name>/`, rebuilt from scratch each
84
- time. the skill under test is installed where the harness discovers skills —
83
+ Per run, under `<root>/.skillcheck/scratch/<name>/`, rebuilt from scratch each
84
+ time. The skill under test is installed where the harness discovers skills:
85
85
  `.claude/skills/<skill>/`, plus `.agents/skills/<skill>/` on codex, or
86
- `.cursor/skills/<skill>/` alone on cursor — with its `evals/` directory
86
+ `.cursor/skills/<skill>/` alone on cursor, with its `evals/` directory
87
87
  excluded, so criteria never leak into the agent's context.
88
88
 
89
- scenario quality is behavioral proof; [authoring](authoring.md) covers the
89
+ Scenario quality is behavioral proof; [authoring](authoring.md) covers the
90
90
  judgment layer lint and evals cannot grade.
package/docs/usage.md CHANGED
@@ -1,60 +1,62 @@
1
- # usage
1
+ # Usage
2
2
 
3
- every subcommand resolves one root, `--root <dir>` or the current directory.
3
+ Every subcommand resolves one root, `--root <dir>` or the current directory.
4
4
  `lint` also takes the root as a positional, because that is the shape CI reaches
5
5
  for first.
6
6
 
7
- ## lint
7
+ ## Lint
8
8
 
9
9
  ```sh
10
10
  skillcheck lint # lints the current repo
11
11
  skillcheck lint ../other # lints another root
12
12
  ```
13
13
 
14
- checks each `<root>/skills/<skill>/`:
14
+ Checks each `<root>/skills/<skill>/`:
15
15
 
16
- - frontmatter opens with `---` on line 1 and closes
17
- - keys are `name`, `description`, `disable-model-invocation` and nothing else,
16
+ - Frontmatter opens with `---` on line 1 and closes
17
+ - Keys are `name`, `description`, `disable-model-invocation` and nothing else,
18
18
  each at most once
19
19
  - `name` equals the directory name; `description` is non-empty
20
20
  - `disable-model-invocation`, when present, is the bare YAML boolean `true`.
21
- a quoted `"true"` is an error
22
- - relative links in the body resolve on disk
21
+ A quoted `"true"` is an error
22
+ - Relative links in the body resolve on disk
23
23
 
24
- code spans and fenced blocks are stripped before links are checked, so example
25
- links never fail. external schemes and `#anchors` pass. dot-directories under
24
+ Code spans and fenced blocks are stripped before links are checked, so example
25
+ links never fail. External schemes and `#anchors` pass. Dot-directories under
26
26
  `skills/` (`.claude-plugin`) are plugin metadata, not packages, and are skipped.
27
27
 
28
- findings print one per line, relative to the linted root, then a count. exit 0
28
+ Findings print one per line, relative to the linted root, then a count. Exit 0
29
29
  clean, 1 with findings.
30
30
 
31
- ## run
31
+ ## Run
32
32
 
33
33
  ```sh
34
34
  skillcheck run skills/<skill>/evals/<scenario>
35
35
  skillcheck run <scenario-dir> --agent MODEL --judge MODEL --harness codex --max-turns 80
36
36
  ```
37
37
 
38
- materializes the scenario into `<root>/.skillcheck/scratch/<name>/workdir`,
38
+ Materializes the scenario into `<root>/.skillcheck/scratch/<name>/workdir`,
39
39
  installs the skill under test into that workdir, drives the agent, and grades
40
- the files it wrote. exit 0 pass, 1 graded fail, 2 error (promptfoo produced no
41
- usable result).
40
+ the files it wrote. Exit 0 means pass, 1 means graded fail, and 2 means error.
41
+ Exit 2 covers missing usable promptfoo output or optional eval peers. The
42
+ message carries the exact `pnpm add` command; see
43
+ [adoption](adoption.md#evals).
42
44
 
43
- a test that errored was never graded, so it exits 2, prints the provider's
44
- message, and writes no provenance sidecar. it is never reported as
45
+ A test that errored was never graded, so it exits 2, prints the provider's
46
+ message, and writes no provenance sidecar. It is never reported as
45
47
  `FAIL score=0.0000`; only a real judged verdict can fail a run.
46
48
 
47
- defaults: `--harness claude`, agent `claude-opus-5`, judge `claude-opus-5`,
48
- `--max-turns 50`. on the codex and cursor harnesses, omitting `--agent` leaves
49
+ Defaults: `--harness claude`, agent `claude-opus-5`, judge `claude-opus-5`,
50
+ `--max-turns 50`. On the codex and cursor harnesses, omitting `--agent` leaves
49
51
  the model to that CLI's own default.
50
52
 
51
53
  `--harness cursor` drives the scenario through the Cursor Agent CLI
52
54
  (`cursor-agent` on PATH) with the skill installed under `.cursor/skills/`;
53
- `--agent` names a Cursor model id, e.g. `composer-2.5`. there is no promptfoo
55
+ `--agent` names a Cursor model id, e.g. `composer-2.5`. There is no promptfoo
54
56
  cursor provider, so the run uses this package's own provider module, which
55
57
  replays the CLI's `stream-json` output: the `result` event becomes the graded
56
58
  output and `SKILL.md` reads under `.cursor/skills/` become the `skill-used`
57
- evidence. the judge leg is unchanged.
59
+ evidence. The judge leg is unchanged.
58
60
 
59
61
  `--judge` takes either a bare Claude model (graded through the Anthropic
60
62
  selection in [auth](#auth)) or a provider-qualified promptfoo id, passed
@@ -64,55 +66,63 @@ through verbatim:
64
66
  skillcheck run <scenario-dir> --judge openai:chat:gpt-5.6-sol --judge-effort high
65
67
  ```
66
68
 
67
- a provider-qualified judge authenticates through that provider's own env
69
+ A provider-qualified judge authenticates through that provider's own env
68
70
  (`OPENAI_API_KEY`, plus `OPENAI_BASE_URL` for a gateway) and is recorded
69
71
  verbatim in the scorecard's `judge_model` column. `--judge-effort`
70
72
  (minimal|low|medium|high) sets `reasoning_effort` and requires a
71
73
  provider-qualified judge; the Anthropic judge does not take one.
72
74
 
73
- ## sweep
75
+ ## Sweep
74
76
 
75
77
  ```sh
76
78
  skillcheck sweep # only scenarios without results
77
79
  skillcheck sweep --all # rerun everything
78
80
  ```
79
81
 
80
- walks `<root>/skills/*/evals/*` and `<root>/cli/*/skills/*/evals/*`, in sorted
81
- order, sequentially. a scenario needs both `task.md` and `criteria.json` to be
82
- discovered. exit 2 if anything errored, 1 if anything failed, else 0.
82
+ Walks `<root>/skills/*/evals/*` and `<root>/cli/*/skills/*/evals/*`, in sorted
83
+ order, sequentially. A scenario needs both `task.md` and `criteria.json` to be
84
+ discovered. Exit 2 if anything errored, 1 if anything failed, else 0.
83
85
 
84
- `EVALS_CONCURRENCY` is passed to promptfoo as `-j` (default 4). it parallelizes
86
+ `EVALS_CONCURRENCY` is passed to promptfoo as `-j` (default 4). It parallelizes
85
87
  within one scenario, not across them.
86
88
 
87
- one known failure mode: judge calls through a gateway can drop at the transport
88
- layer ([uinaf/agent-platform#28](https://github.com/uinaf/agent-platform/issues/28)).
89
- that surfaces as an ERROR with no usable result, not as a graded FAIL, and the
89
+ One known failure mode: judge calls through a gateway can drop at the transport
90
+ layer ([uinaf/zebroid-infra#44](https://github.com/uinaf/zebroid-infra/issues/44)).
91
+ That surfaces as an ERROR with no usable result, not as a graded FAIL, and the
90
92
  mitigation is a rerun. `sweep` without `--all` resumes, so a rerun only picks up
91
93
  what is missing.
92
94
 
93
- ## summarize
95
+ ## Summarize
94
96
 
95
97
  ```sh
96
98
  skillcheck summarize [--allow-mixed]
97
99
  ```
98
100
 
99
- reduces `<root>/.skillcheck/results/*.json` into
101
+ Reduces `<root>/.skillcheck/results/*.json` into
100
102
  `<root>/.skillcheck/scorecards/<UTC-date>.json`: one entry per scenario with
101
103
  skill, scenario, harness, tree sha, score, pass, both models, latency, tokens.
102
104
 
103
- if a scorecard for today already exists, the two are merged on
105
+ If a scorecard for today already exists, the two are merged on
104
106
  `(skill, scenario, harness)`: entries from this run win, entries it did not
105
- touch survive, and the merge is reported on stdout. summarizing after rerunning
107
+ touch survive, and the merge is reported on stdout. Summarizing after rerunning
106
108
  six of twenty-nine scenarios therefore leaves twenty-nine rows in the file, not
107
- six. a same-date file that cannot be parsed stops the write instead of being
109
+ six. A same-date file that cannot be parsed stops the write instead of being
108
110
  overwritten.
109
111
 
110
- files that are not promptfoo results are skipped with a warning rather than
111
- failing the reduction.
112
+ Files that are not promptfoo results and ungraded transport errors are skipped
113
+ with a warning rather than failing the reduction. Graded assertion failures
114
+ remain scored results. If a skipped file matches an existing scorecard row,
115
+ summary generation fails and leaves the scorecard unchanged, so an errored rerun
116
+ cannot carry forward its old score. This also applies with `--allow-mixed`.
117
+ Runs keep a `<name>.json.attempt` marker until a graded result and its provenance
118
+ are written. An outstanding marker makes `summarize` skip that identity even
119
+ when the child produced no result file or left partial output. The marker does
120
+ not count as a result for the sweep's existence check, so no-output failures
121
+ remain eligible for retry.
112
122
 
113
- ## provenance
123
+ ## Provenance
114
124
 
115
- each successful run writes a `<name>.meta.json` sidecar next to its result:
125
+ Each successful run writes a `<name>.meta.json` sidecar next to its result:
116
126
 
117
127
  ```json
118
128
  {
@@ -124,27 +134,28 @@ each successful run writes a `<name>.meta.json` sidecar next to its result:
124
134
  ```
125
135
 
126
136
  `summarize` reads those sidecars and refuses to mix skills-tree revisions in one
127
- scorecard unless `--allow-mixed`, in which case the top-level `skills_tree_sha`
128
- becomes `mixed` and per-entry shas remain. a result with no sidecar reduces as
137
+ scorecard, including retained rows from partial reruns, unless `--allow-mixed`.
138
+ Rejection leaves the existing scorecard unchanged. With the override, the top-level `skills_tree_sha`
139
+ becomes `mixed` and per-entry shas remain. A result with no sidecar reduces as
129
140
  `unattested`.
130
141
 
131
- ## state
142
+ ## State
132
143
 
133
144
  `<root>/.skillcheck/` holds `scratch/` and `results/`, both disposable and safe
134
- to gitignore, and `scorecards/`, which is meant to be committed. nothing is ever
145
+ to gitignore, and `scorecards/`, which is meant to be committed. Nothing is ever
135
146
  written inside the installed package.
136
147
 
137
- ## auth
148
+ ## Auth
138
149
 
139
- | variable | effect |
150
+ | Variable | Effect |
140
151
  | --------------------------------------------- | ----------------------------------------------------------------- |
141
- | `ANTHROPIC_BASE_URL` + `ANTHROPIC_AUTH_TOKEN` | claude agent and judge go through a gateway |
142
- | none of the above | falls back to the local Claude Code session |
143
- | `ANTHROPIC_API_KEY` | judge grades over `anthropic:messages:<model>` instead of the SDK |
144
- | `CODEX_HOME` (default `~/.codex`) | where the codex harness finds the local `codex` CLI login |
145
- | `OPENAI_API_KEY` | codex agent auth when there is no local login |
146
- | `CURSOR_API_KEY` | cursor agent auth; a logged-in `cursor-agent` also works |
147
- | `OPENAI_API_KEY` + `OPENAI_BASE_URL` | a provider-qualified `--judge openai:…`, optionally via a gateway |
148
-
149
- a bare `--judge` model stays on the Anthropic selection regardless of the
152
+ | `ANTHROPIC_BASE_URL` + `ANTHROPIC_AUTH_TOKEN` | The claude agent and judge go through a gateway |
153
+ | None of the above | Falls back to the local Claude Code session |
154
+ | `ANTHROPIC_API_KEY` | Judge grades over `anthropic:messages:<model>` instead of the SDK |
155
+ | `CODEX_HOME` (default `~/.codex`) | Where the codex harness finds the local `codex` CLI login |
156
+ | `OPENAI_API_KEY` | Agent auth for codex when there is no local login |
157
+ | `CURSOR_API_KEY` | Agent auth for cursor; a logged-in `cursor-agent` also works |
158
+ | `OPENAI_API_KEY` + `OPENAI_BASE_URL` | A provider-qualified `--judge openai:…`, optionally via a gateway |
159
+
160
+ A bare `--judge` model stays on the Anthropic selection regardless of the
150
161
  agent harness; a provider-qualified `--judge` uses that provider's env instead.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "0.4.0",
3
+ "version": "0.5.1",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {
@@ -27,22 +27,37 @@
27
27
  "registry": "https://registry.npmjs.org/"
28
28
  },
29
29
  "scripts": {
30
- "verify": "vp check && vp pack && vp test run",
30
+ "verify": "vp run ready",
31
+ "verify:full": "vp run --no-cache ready",
31
32
  "prepare": "vp config --no-agent",
32
- "prepublishOnly": "pnpm run verify"
33
+ "prepublishOnly": "pnpm run verify:full"
33
34
  },
34
- "dependencies": {
35
+ "devDependencies": {
35
36
  "@anthropic-ai/claude-agent-sdk": "^0.3.233",
36
37
  "@openai/codex-sdk": "^0.147.0",
37
- "promptfoo": "^0.122.0"
38
- },
39
- "devDependencies": {
40
38
  "@types/node": "^26.2.0",
39
+ "promptfoo": "^0.122.0",
41
40
  "vite": "catalog:",
42
41
  "vite-plus": "catalog:"
43
42
  },
43
+ "peerDependencies": {
44
+ "@anthropic-ai/claude-agent-sdk": "^0.3.233",
45
+ "@openai/codex-sdk": "^0.147.0",
46
+ "promptfoo": "^0.122.0"
47
+ },
48
+ "peerDependenciesMeta": {
49
+ "@anthropic-ai/claude-agent-sdk": {
50
+ "optional": true
51
+ },
52
+ "@openai/codex-sdk": {
53
+ "optional": true
54
+ },
55
+ "promptfoo": {
56
+ "optional": true
57
+ }
58
+ },
44
59
  "engines": {
45
60
  "node": ">=24"
46
61
  },
47
- "packageManager": "pnpm@11.22.0"
62
+ "packageManager": "pnpm@12.0.0"
48
63
  }