@uinaf/skillcheck 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -2,25 +2,28 @@
2
2
 
3
3
  # uinaf/skillcheck
4
4
 
5
- lint and eval harness for agent skills. one CLI with two halves: a keyless
5
+ Lint and eval harness for agent skills. One CLI with two halves: a keyless
6
6
  structural lint any repo can run in CI, and a promptfoo-driven eval loop that
7
7
  grades what a skill actually makes an agent do.
8
8
 
9
- built for [uinaf](https://uinaf.dev) skill repos. nothing in it is
10
- uinaf-specific. it ships no opinion about what a skill should say, only about
9
+ Built for [uinaf](https://uinaf.dev) skill repos. Nothing in it is
10
+ uinaf-specific. It ships no opinion about what a skill should say, only about
11
11
  where skills sit and how a scenario is scored.
12
12
 
13
- ## install
13
+ ## Install
14
14
 
15
15
  ```sh
16
16
  pnpm add -D @uinaf/skillcheck
17
17
  ```
18
18
 
19
- node 24 or newer. the package ships compiled ESM and runs no install scripts,
20
- so a runner using `--ignore-scripts` is fine. consumers still pinned to the
21
- pre-npm git tags are covered in [adoption](docs/adoption.md).
19
+ Node 24 or newer. The package ships compiled ESM, runs no install scripts, and
20
+ has no regular dependencies, so a runner using `--ignore-scripts` is fine and
21
+ the lint-only install stays at a handful of packages. The promptfoo eval engine
22
+ and provider SDKs are optional peers, installed only on the operator machine
23
+ that runs evals ([adoption](docs/adoption.md)). Consumers still pinned to the
24
+ pre-npm git tags are covered there too.
22
25
 
23
- ## use
26
+ ## Use
24
27
 
25
28
  ```sh
26
29
  skillcheck lint # structural lint, no credentials
@@ -29,12 +32,12 @@ skillcheck sweep && skillcheck summarize # every scenario, then a scorecard
29
32
  ```
30
33
 
31
34
  `lint` needs nothing. `run` and `sweep` need model auth, which is why sweeps
32
- stay operator-run and consumer repos never hold credentials. every command,
35
+ stay operator-run and consumer repos never hold credentials. Every command,
33
36
  flag, and auth variable is in [usage](docs/usage.md).
34
37
 
35
- ## layout contract
38
+ ## Layout contract
36
39
 
37
- frozen, not configurable. every command reads one root: `--root <dir>`, or the
40
+ Frozen, not configurable. Every command reads one root: `--root <dir>`, or the
38
41
  current directory.
39
42
 
40
43
  ```text
@@ -47,18 +50,18 @@ current directory.
47
50
  `cli/*/skills/<skill>/` is scanned too, for repos that keep a skill next to the
48
51
  CLI it documents.
49
52
 
50
- ## docs
53
+ ## Docs
51
54
 
52
- | doc | when |
55
+ | Doc | When |
53
56
  | --------------------------------------------------------------- | ------------------------------------- |
54
- | [usage](docs/usage.md) | every subcommand, flag, and auth path |
55
- | [scenarios](docs/scenarios.md) | writing an eval scenario |
56
- | [authoring](docs/authoring.md) | writing and auditing the skill itself |
57
- | [adoption](docs/adoption.md) | wiring the lint into another repo |
58
- | [releasing](docs/releasing.md) | the npm pipeline |
59
- | [contributing](CONTRIBUTING.md) | local setup and the verify gate |
60
- | [security](https://github.com/uinaf/skillcheck/security/policy) | reporting a vulnerability |
61
-
62
- ## license
57
+ | [Usage](docs/usage.md) | Every subcommand, flag, and auth path |
58
+ | [Scenarios](docs/scenarios.md) | Writing an eval scenario |
59
+ | [Authoring](docs/authoring.md) | Writing and auditing the skill itself |
60
+ | [Adoption](docs/adoption.md) | Wiring the lint into another repo |
61
+ | [Releasing](docs/releasing.md) | The npm pipeline |
62
+ | [Contributing](CONTRIBUTING.md) | Local setup and the verify gate |
63
+ | [Security](https://github.com/uinaf/skillcheck/security/policy) | Reporting a vulnerability |
64
+
65
+ ## License
63
66
 
64
67
  MIT · undefined is not a function LLC
package/dist/cli.js CHANGED
@@ -1,6 +1,6 @@
1
1
  #!/usr/bin/env node
2
2
  import { lintSkills } from "./lint.js";
3
- import { generateRun, runNameFor } from "./scenario.js";
3
+ import { generateRun, requiredEvalPackages, resolvePackageDir, runNameFor } from "./scenario.js";
4
4
  import { execFileSync, spawnSync } from "node:child_process";
5
5
  import fs from "node:fs";
6
6
  import path from "node:path";
@@ -81,6 +81,29 @@ function runOptions(flags) {
81
81
  maxTurns: flags.has("--max-turns") ? parseMaxTurns(flags.get("--max-turns")) : void 0
82
82
  };
83
83
  }
84
+ function ensureEvalPackages(opts) {
85
+ const missing = requiredEvalPackages(opts, process.env.ANTHROPIC_API_KEY !== void 0).filter((pkg) => resolvePackageDir(pkg) === void 0);
86
+ if (missing.length === 0) return;
87
+ const peers = JSON.parse(fs.readFileSync(path.join(packageDir, "package.json"), "utf8")).peerDependencies ?? {};
88
+ const specs = missing.map((pkg) => `"${pkg}@${peers[pkg] ?? "latest"}"`).join(" ");
89
+ console.error([
90
+ `missing eval package(s): ${missing.join(", ")}`,
91
+ "",
92
+ "The eval engine is an optional peer so `skillcheck lint` installs stay",
93
+ "small. Evals are operator-run; install the peers next to @uinaf/skillcheck:",
94
+ "",
95
+ ` pnpm add -D ${specs}`
96
+ ].join("\n"));
97
+ process.exit(2);
98
+ }
99
+ function promptfooEntry() {
100
+ const dir = resolvePackageDir("promptfoo");
101
+ if (dir === void 0) throw new Error("promptfoo is not installed");
102
+ const bin = JSON.parse(fs.readFileSync(path.join(dir, "package.json"), "utf8")).bin;
103
+ const rel = typeof bin === "string" ? bin : bin?.promptfoo;
104
+ if (typeof rel !== "string") throw new Error(`promptfoo at ${dir} declares no bin`);
105
+ return path.join(dir, rel);
106
+ }
84
107
  function classifyResult(raw) {
85
108
  const root = raw;
86
109
  const res = root?.results?.results?.[0];
@@ -117,8 +140,8 @@ function runScenario(scenarioDir, opts, root) {
117
140
  fs.rmSync(resultPath, { force: true });
118
141
  fs.rmSync(metaPath(resultPath), { force: true });
119
142
  const sha = gitHead(root);
120
- const rc = spawnSync("npx", [
121
- "promptfoo",
143
+ const rc = spawnSync(process.execPath, [
144
+ promptfooEntry(),
122
145
  "eval",
123
146
  "--no-cache",
124
147
  "--no-progress-bar",
@@ -182,7 +205,9 @@ function discoverScenarios(root) {
182
205
  function cmdRun(argv) {
183
206
  const { positional, flags } = parseArgs(argv);
184
207
  if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--judge-effort EFFORT] [--harness claude|codex|cursor]");
185
- const o = runScenario(positional[0], runOptions(flags), resolveRoot(flags));
208
+ const opts = runOptions(flags);
209
+ ensureEvalPackages(opts);
210
+ const o = runScenario(positional[0], opts, resolveRoot(flags));
186
211
  if (o.score === void 0) {
187
212
  console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
188
213
  process.exit(2);
@@ -195,6 +220,7 @@ function cmdSweep(argv) {
195
220
  if (positional.length > 0) fail("usage: skillcheck sweep [--root DIR] [--all]");
196
221
  const root = resolveRoot(flags);
197
222
  const opts = runOptions(flags);
223
+ ensureEvalPackages(opts);
198
224
  const all = flags.get("--all") === true;
199
225
  const resultsDir = stateDirs(root).results;
200
226
  let passed = 0, failed = 0, errored = 0, skipped = 0;
package/dist/scenario.js CHANGED
@@ -213,8 +213,25 @@ function buildConfig(s, workdir, manifestPath, opts, paths) {
213
213
  }
214
214
  const SDK_PACKAGES = ["@anthropic-ai/claude-agent-sdk", "@openai/codex-sdk"];
215
215
  function sdkNodeModulesDir() {
216
+ for (const pkg of SDK_PACKAGES) {
217
+ const dir = holdingNodeModules(pkg);
218
+ if (dir !== void 0) return dir;
219
+ }
220
+ }
221
+ function holdingNodeModules(pkg) {
216
222
  const require = createRequire(import.meta.url);
217
- for (const pkg of SDK_PACKAGES) for (const dir of require.resolve.paths(pkg) ?? []) if (fs.existsSync(path.join(dir, pkg, "package.json"))) return dir;
223
+ for (const dir of require.resolve.paths(pkg) ?? []) if (fs.existsSync(path.join(dir, pkg, "package.json"))) return dir;
224
+ }
225
+ function resolvePackageDir(pkg) {
226
+ const dir = holdingNodeModules(pkg);
227
+ return dir === void 0 ? void 0 : path.join(dir, pkg);
228
+ }
229
+ function requiredEvalPackages(opts, hasAnthropicKey) {
230
+ const pkgs = ["promptfoo"];
231
+ const sdkJudge = !opts.judgeModel.includes(":") && !hasAnthropicKey;
232
+ if (opts.harness === "claude" || sdkJudge) pkgs.push("@anthropic-ai/claude-agent-sdk");
233
+ if (opts.harness === "codex") pkgs.push("@openai/codex-sdk");
234
+ return pkgs;
218
235
  }
219
236
  function generateRun(scenarioDir, opts, paths) {
220
237
  const s = loadScenario(scenarioDir);
@@ -239,4 +256,4 @@ function generateRun(scenarioDir, opts, paths) {
239
256
  };
240
257
  }
241
258
  //#endregion
242
- export { buildConfig, generateRun, isHiddenSkill, loadScenario, materialize, runNameFor, sdkNodeModulesDir, stripHiddenFlag };
259
+ export { buildConfig, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag };
package/docs/adoption.md CHANGED
@@ -1,15 +1,15 @@
1
- # adopting it in a repo
1
+ # Adopting it in a repo
2
2
 
3
- two separate decisions: run the lint in CI, and run evals on a machine that has
4
- model auth. only the first belongs in a consumer repo.
3
+ Two separate decisions: run the lint in CI, and run evals on a machine that has
4
+ model auth. Only the first belongs in a consumer repo.
5
5
 
6
- ## lint in CI
6
+ ## Lint in CI
7
7
 
8
8
  ```sh
9
9
  pnpm add -D @uinaf/skillcheck
10
10
  ```
11
11
 
12
- runners that install with `--ignore-scripts` are fine: the package ships
12
+ Runners that install with `--ignore-scripts` are fine: the package ships
13
13
  compiled ESM and has no install, prepare, or postinstall script.
14
14
 
15
15
  ```json
@@ -32,49 +32,59 @@ jobs:
32
32
  - run: pnpm run skills:lint
33
33
  ```
34
34
 
35
- the job needs no secrets and no network beyond the install. node 24 is the
35
+ The job needs no secrets and no network beyond the install. Node 24 is the
36
36
  floor.
37
37
 
38
- run it through the script rather than a bare `npx skillcheck`: the script
38
+ Run it through the script rather than a bare `npx skillcheck`: the script
39
39
  resolves the version the repo pinned, and `npx` would resolve the latest one on
40
40
  the registry.
41
41
 
42
- ## evals
42
+ ## Evals
43
43
 
44
- sweeps need model credentials, so they stay off consumer CI and run from an
45
- operator machine or a job that already holds gateway auth:
44
+ Sweeps need model credentials, so they stay off consumer CI and run from an
45
+ operator machine or a job that already holds gateway auth. They also need the
46
+ eval engine, which is an optional peer precisely so the lint-only install
47
+ above stays small — install it next to the package on the operator machine:
48
+
49
+ ```sh
50
+ pnpm add -D promptfoo @anthropic-ai/claude-agent-sdk @openai/codex-sdk
51
+ ```
52
+
53
+ `run` and `sweep` check for the peers the selected harness needs and exit 2
54
+ with that install command when they are missing, so a lint-only install never
55
+ crashes into a resolution error. Then:
46
56
 
47
57
  ```sh
48
58
  skillcheck sweep # resumes: only scenarios without results
49
59
  skillcheck summarize # writes .skillcheck/scorecards/<UTC-date>.json
50
60
  ```
51
61
 
52
- commit `.skillcheck/scorecards/`. gitignore the rest:
62
+ Commit `.skillcheck/scorecards/`. Gitignore the rest:
53
63
 
54
64
  ```gitignore
55
65
  .skillcheck/results/
56
66
  .skillcheck/scratch/
57
67
  ```
58
68
 
59
- a scorecard is only comparable against the tree it graded, which is why every
69
+ A scorecard is only comparable against the tree it graded, which is why every
60
70
  result carries the root repo's HEAD and `summarize` refuses to mix revisions
61
71
  without `--allow-mixed`.
62
72
 
63
- ## upgrading
73
+ ## Upgrading
64
74
 
65
- bump the version in `package.json` and rerun the sweep. results carry the
75
+ Bump the version in `package.json` and rerun the sweep. Results carry the
66
76
  `tool_version` that produced them, so a scorecard says which harness build it
67
77
  came from as well as which skills tree.
68
78
 
69
- ## the older install path
79
+ ## The older install path
70
80
 
71
- before the package was published, consumers installed it from a git tag:
81
+ Before the package was published, consumers installed it from a git tag:
72
82
 
73
83
  ```sh
74
84
  npm i -D github:uinaf/skillcheck#v0.1.3
75
85
  ```
76
86
 
77
- those tags are frozen and still work: they carry a committed `dist/`, and npm
78
- 12 consumers needed `allow-git=root` in `.npmrc` to accept the spec at all. the
79
- registry install needs none of that. tags from `v0.1.4` on are npm releases and
87
+ Those tags are frozen and still work: they carry a committed `dist/`, and npm
88
+ 12 consumers needed `allow-git=root` in `.npmrc` to accept the spec at all. The
89
+ registry install needs none of that. Tags from `v0.1.4` on are npm releases and
80
90
  carry no `dist/`, so a git spec pointing at one will not run.
package/docs/authoring.md CHANGED
@@ -1,70 +1,70 @@
1
- # authoring and auditing skills
1
+ # Authoring and auditing skills
2
2
 
3
- what `lint` and evals cannot judge: whether a skill is worth routing to and
4
- cheap to load. use this when writing a skill or auditing one. evidence beats
3
+ What `lint` and evals cannot judge: whether a skill is worth routing to and
4
+ cheap to load. Use this when writing a skill or auditing one. Evidence beats
5
5
  stylistic preference; run `skillcheck lint` first and let this cover the rest.
6
6
 
7
- ## metadata and discovery
7
+ ## Metadata and discovery
8
8
 
9
9
  - `name` is concrete and easy to say out loud. `helper`, `tools`, `utils` are
10
10
  discovery smells.
11
11
  - `description` is third person and says both what the skill does and when to
12
- use it. it is an always-loaded retrieval pointer: front-load the concrete
12
+ use it. It is an always-loaded retrieval pointer: front-load the concrete
13
13
  action or domain that should activate it.
14
- - one trigger per materially distinct request branch. collapse synonyms that
14
+ - One trigger per materially distinct request branch. Collapse synonyms that
15
15
  rename the same branch.
16
- - state the main overlap boundary without naming another skill.
16
+ - State the main overlap boundary without naming another skill.
17
17
 
18
- ## body shape
18
+ ## Body shape
19
19
 
20
- - keep `SKILL.md` on workflow, principles, boundaries, and routing. lead with
20
+ - Keep `SKILL.md` on workflow, principles, boundaries, and routing. Lead with
21
21
  the task, not a bibliography.
22
- - assume the model is smart; spend tokens on repo-specific judgment. delete any
22
+ - Assume the model is smart; spend tokens on repo-specific judgment. Delete any
23
23
  instruction that would not change a capable model's behavior.
24
- - match freedom to risk: high for contextual judgment, medium when a preferred
24
+ - Match freedom to risk: high for contextual judgment, medium when a preferred
25
25
  pattern exists, low for fragile operations.
26
- - say what evidence to gather and what a complete result includes. end each
26
+ - Say what evidence to gather and what a complete result includes. End each
27
27
  step with an observable completion condition, not "understood" or "handled".
28
28
 
29
- ## progressive disclosure
29
+ ## Progressive disclosure
30
30
 
31
- - durable detail, rubrics, and long examples go in `references/`, one hop from
32
- `SKILL.md`, each with a task-shaped retrieval job. material every path needs
31
+ - Durable detail, rubrics, and long examples go in `references/`, one hop from
32
+ `SKILL.md`, each with a task-shaped retrieval job. Material every path needs
33
33
  stays inline.
34
- - for repeated deterministic work, route to the target's existing framework,
34
+ - For repeated deterministic work, route to the target's existing framework,
35
35
  schema, task graph, or library; otherwise add a tested module in the
36
36
  project's primary language, not ad-hoc shell rendered as prose.
37
- - when executable code belongs to another maintained project, link the exact
37
+ - When executable code belongs to another maintained project, link the exact
38
38
  public artifact and state the contract it demonstrates; do not fork it into
39
39
  prose.
40
- - a package stays independently usable: state prerequisites and out-of-scope
41
- next steps locally. never invoke, import, or assume a sibling skill.
40
+ - A package stays independently usable: state prerequisites and out-of-scope
41
+ next steps locally. Never invoke, import, or assume a sibling skill.
42
42
 
43
- ## audit
43
+ ## Audit
44
44
 
45
- grade each dimension strong, mixed, or weak:
45
+ Grade each dimension strong, mixed, or weak:
46
46
 
47
- | dimension | question |
47
+ | Dimension | Question |
48
48
  | ---------------------- | --------------------------------------------------------------------- |
49
- | discovery | does metadata alone route a realistic request here |
50
- | workflow | does the body say how to begin, what evidence to gather, when to stop |
51
- | progressive disclosure | is detail in the right file |
52
- | repo fit | are links, commands, and conventions current |
53
- | verification | is the strongest mechanical check named, plus a real evidence loop |
54
- | boundaries | are limits and next steps stated without leaning on a sibling skill |
49
+ | Discovery | Does metadata alone route a realistic request here |
50
+ | Workflow | Does the body say how to begin, what evidence to gather, when to stop |
51
+ | Progressive disclosure | Is detail in the right file |
52
+ | Repo fit | Are links, commands, and conventions current |
53
+ | Verification | Is the strongest mechanical check named, plus a real evidence loop |
54
+ | Boundaries | Are limits and next steps stated without leaning on a sibling skill |
55
55
 
56
- blockers, must-fix: invalid frontmatter; a description that fails discovery;
56
+ Blockers, must-fix: invalid frontmatter; a description that fails discovery;
57
57
  stale commands, paths, or links; a workflow with no start, evidence loop, or
58
58
  completion; conflicts with the repo's guidance; sibling-skill dependencies.
59
59
 
60
- major findings: vague name; synonym-stuffed description; bloated `SKILL.md`;
60
+ Major findings: vague name; synonym-stuffed description; bloated `SKILL.md`;
61
61
  missing or muddy boundaries; prose re-inventing a deterministic tool; abstract
62
62
  examples.
63
63
 
64
- ## improve
64
+ ## Improve
65
65
 
66
- fix blockers first, then the highest-leverage majors. prefer the smallest
67
- change that improves activation, decision quality, or proof. when pruning,
66
+ Fix blockers first, then the highest-leverage majors. Prefer the smallest
67
+ change that improves activation, decision quality, or proof. When pruning,
68
68
  measure common-path context for representative requests; line count alone does
69
- not reveal retrieval cost. after edits, rerun `skillcheck lint` and the repo's
69
+ not reveal retrieval cost. After edits, rerun `skillcheck lint` and the repo's
70
70
  gate, and rerun evals when behavior was the thing changed.
package/docs/releasing.md CHANGED
@@ -1,8 +1,8 @@
1
- # releasing
1
+ # Releasing
2
2
 
3
- ## pipeline
3
+ ## Pipeline
4
4
 
5
- a push to `main` runs one workflow, `.github/workflows/release.yml`:
5
+ A push to `main` runs one workflow, `.github/workflows/release.yml`:
6
6
 
7
7
  ```text
8
8
  verify ──┐
@@ -12,69 +12,69 @@ scan ────┘
12
12
 
13
13
  `verify` and `scan` are the shared gate: `verify` is called from `verify.yml`,
14
14
  and `scan` calls the shared scan in `uinaf/.github`, the same one `scan.yml`
15
- runs for pull requests. keep it that way: a second copy of the gate on a push-to-`main` workflow
15
+ runs for pull requests. Keep it that way: a second copy of the gate on a push-to-`main` workflow
16
16
  races this one over the same commit.
17
17
 
18
- the file name `release.yml` is load-bearing. see below.
18
+ The file name `release.yml` is load-bearing. See below.
19
19
 
20
20
  ## npm
21
21
 
22
22
  `@uinaf/skillcheck` publishes from `.github/workflows/release.yml` via npm
23
- Trusted Publishing (OIDC) and the `uinaf-releaser` GitHub App. there is no npm
23
+ Trusted Publishing (OIDC) and the `uinaf-releaser` GitHub App. There is no npm
24
24
  token in this repository, in its environments, or in the organization.
25
25
 
26
- required on the `release` GitHub Environment:
26
+ Required on the `release` GitHub Environment:
27
27
 
28
- | name | kind | purpose |
28
+ | Name | Kind | Purpose |
29
29
  | ------------------------------- | ------ | ------------------------------------------- |
30
30
  | `UINAF_RELEASE_APP_CLIENT_ID` | var | GitHub App client id for the releaser bot |
31
31
  | `UINAF_RELEASE_APP_PRIVATE_KEY` | secret | GitHub App private key for the releaser bot |
32
32
 
33
- the trusted publisher on npmjs.com is registered by **file path**, so
33
+ The trusted publisher on npmjs.com is registered by **file path**, so
34
34
  `.github/workflows/release.yml` cannot be renamed or moved without editing that
35
- registration first. a rename fails the publish with an identity mismatch, and
36
- nothing earlier in the run reports it. the `release` environment name is bound
35
+ registration first. A rename fails the publish with an identity mismatch, and
36
+ nothing earlier in the run reports it. The `release` environment name is bound
37
37
  the same way.
38
38
 
39
- deleting the `release` environment deletes both rows above with it, and there is
39
+ Deleting the `release` environment deletes both rows above with it, and there is
40
40
  no repo-level fallback: `create-github-app-token` then runs with empty inputs
41
- and the job fails at that step. the private key cannot be read back from
41
+ and the job fails at that step. The private key cannot be read back from
42
42
  GitHub; recreating it means generating a new one in the App settings.
43
43
 
44
- ## version history
44
+ ## Version history
45
45
 
46
- semantic-release owns the version and the tag. `tagFormat` is `v${version}` and
46
+ The version and the tag are owned by semantic-release. `tagFormat` is `v${version}` and
47
47
  history continues from `v0.1.3`; `v0.1.0`–`v0.1.3` are the legacy git-install
48
48
  tags and are never deleted or moved.
49
49
 
50
- during preparation, `@semantic-release/npm` stages the released `package.json`
50
+ During preparation, `@semantic-release/npm` stages the released `package.json`
51
51
  version and `@jno21/semantic-release-github-commit` commits it to `main` through
52
52
  GitHub's API as the authenticated App. GitHub signs that commit, and the release
53
- tag points to it. the `[skip ci]` marker on that commit is what stops a release
53
+ tag points to it. The `[skip ci]` marker on that commit is what stops a release
54
54
  from releasing itself.
55
55
 
56
- check what the next version would be without publishing anything:
56
+ Check what the next version would be without publishing anything:
57
57
 
58
58
  ```sh
59
59
  pnpm dlx semantic-release --dry-run --no-ci
60
60
  ```
61
61
 
62
- ## the artifact
62
+ ## The artifact
63
63
 
64
64
  `dist/` is generated and untracked. `prepublishOnly` runs `pnpm run verify`,
65
65
  which builds it, so the tarball is always packed from a tree that just passed
66
66
  the gate. `files` is `dist`, `docs`, `README.md`, `LICENSE`.
67
67
 
68
- manual publish is emergency recovery only:
68
+ Manual publish is emergency recovery only:
69
69
 
70
70
  ```sh
71
71
  pnpm run verify
72
72
  npm publish --access public
73
73
  ```
74
74
 
75
- ## the old install path
75
+ ## The old install path
76
76
 
77
- before `@uinaf/skillcheck` existed, consumers installed
78
- `github:uinaf/skillcheck#v0.1.3`. those tags still resolve and still carry a
79
- committed `dist/`, so anything pinned to them keeps working untouched. new
77
+ Before `@uinaf/skillcheck` existed, consumers installed
78
+ `github:uinaf/skillcheck#v0.1.3`. Those tags still resolve and still carry a
79
+ committed `dist/`, so anything pinned to them keeps working untouched. New
80
80
  consumers use npm.
package/docs/scenarios.md CHANGED
@@ -1,20 +1,20 @@
1
- # writing scenarios
1
+ # Writing scenarios
2
2
 
3
- a scenario is two files in a frozen location:
3
+ A scenario is two files in a frozen location:
4
4
 
5
5
  ```text
6
6
  <root>/skills/<skill>/evals/<scenario>/task.md
7
7
  <root>/skills/<skill>/evals/<scenario>/criteria.json
8
8
  ```
9
9
 
10
- the path is the identity: `<skill>--<scenario>` names the run, the result file,
11
- and the scorecard entry. on the codex and cursor harnesses the name gains a
10
+ The path is the identity: `<skill>--<scenario>` names the run, the result file,
11
+ and the scorecard entry. On the codex and cursor harnesses the name gains a
12
12
  `--codex` or `--cursor` suffix, so every harness can hold results side by side.
13
- a directory missing either file is not discovered.
13
+ A directory missing either file is not discovered.
14
14
 
15
15
  ## task.md
16
16
 
17
- the prompt handed to the agent, verbatim, with one piece of syntax. input files
17
+ The prompt handed to the agent, verbatim, with one piece of syntax. Input files
18
18
  are embedded inline and materialized into the workdir before the run:
19
19
 
20
20
  ```md
@@ -25,13 +25,13 @@ Fix the failing check in the config below.
25
25
  ======= END FILE =======
26
26
  ```
27
27
 
28
- each block is replaced in the prompt with a pointer ("Input file `config.json`
29
- is available in your working directory.") and written to disk. destinations
28
+ Each block is replaced in the prompt with a pointer ("Input file `config.json`
29
+ is available in your working directory.") and written to disk. Destinations
30
30
  must stay under the workdir, must not collide, and must not target `.claude/`,
31
31
  `.agents/`, or `.cursor/`, since a fixture that writes agent config would be
32
32
  configuring its own examiner.
33
33
 
34
- write the task the way a user would write it. do not name the skill, describe
34
+ Write the task the way a user would write it. Do not name the skill, describe
35
35
  its steps, or hint at the checklist: routing is part of what is being measured.
36
36
 
37
37
  ## criteria.json
@@ -46,45 +46,45 @@ its steps, or hint at the checklist: routing is part of what is being measured.
46
46
  }
47
47
  ```
48
48
 
49
- `type` must be `weighted_checklist` and the checklist must be non-empty. every
49
+ `type` must be `weighted_checklist` and the checklist must be non-empty. Every
50
50
  item needs a non-empty `name` and `description` and a positive `max_score`.
51
51
 
52
- each item becomes one `llm-rubric` assertion weighted by `max_score`, inside an
53
- assert-set with threshold 0.7. a separate `skill-used` assertion sits outside
52
+ Each item becomes one `llm-rubric` assertion weighted by `max_score`, inside an
53
+ assert-set with threshold 0.7. A separate `skill-used` assertion sits outside
54
54
  that aggregate, so a run that produces good output without ever loading the
55
- skill still fails. there is no test-level threshold: both must pass.
55
+ skill still fails. There is no test-level threshold: both must pass.
56
56
 
57
- write descriptions a judge can check against the deliverable: an observable
58
- property, not a feeling. weight the items that would make a reviewer reject the
57
+ Write descriptions a judge can check against the deliverable: an observable
58
+ property, not a feeling. Weight the items that would make a reviewer reject the
59
59
  work.
60
60
 
61
- ## what the judge sees
61
+ ## What the judge sees
62
62
 
63
- the agent's final message, plus every file in the workdir that differs from the
64
- pre-run manifest. unchanged inputs are omitted; deleted inputs, unreadable
63
+ The agent's final message, plus every file in the workdir that differs from the
64
+ pre-run manifest. Unchanged inputs are omitted; deleted inputs, unreadable
65
65
  files, and non-regular files are named rather than read.
66
66
 
67
- sections are sorted by path, each file is capped at 4,000 characters and the
68
- appended total at 24,000, with truncation stated inline. very large outputs make
69
- rubric judges return nothing at all, which is why the caps exist. keep fixtures
67
+ Sections are sorted by path, each file is capped at 4,000 characters and the
68
+ appended total at 24,000, with truncation stated inline. Very large outputs make
69
+ rubric judges return nothing at all, which is why the caps exist. Keep fixtures
70
70
  small enough that the deliverable fits.
71
71
 
72
- ## hidden skills
72
+ ## Hidden skills
73
73
 
74
- a skill with `disable-model-invocation: true` is explicit-invoke-only in
75
- production, which the agent SDK cannot simulate. so the eval copy, never the shipped one,
74
+ A skill with `disable-model-invocation: true` is explicit-invoke-only in
75
+ production, which the agent SDK cannot simulate. So the eval copy, never the shipped one,
76
76
  has the flag stripped, and the task gains a leading
77
- `Use the <skill> skill for this task.` the eval then measures
78
- behavior-when-invoked rather than routing. the flag is only honored inside the
77
+ `Use the <skill> skill for this task.` The eval then measures
78
+ behavior-when-invoked rather than routing. The flag is only honored inside the
79
79
  frontmatter block; body text mentioning the key does not count.
80
80
 
81
- ## the workdir
81
+ ## The workdir
82
82
 
83
- per run, under `<root>/.skillcheck/scratch/<name>/`, rebuilt from scratch each
84
- time. the skill under test is installed where the harness discovers skills —
83
+ Per run, under `<root>/.skillcheck/scratch/<name>/`, rebuilt from scratch each
84
+ time. The skill under test is installed where the harness discovers skills:
85
85
  `.claude/skills/<skill>/`, plus `.agents/skills/<skill>/` on codex, or
86
- `.cursor/skills/<skill>/` alone on cursor — with its `evals/` directory
86
+ `.cursor/skills/<skill>/` alone on cursor, with its `evals/` directory
87
87
  excluded, so criteria never leak into the agent's context.
88
88
 
89
- scenario quality is behavioral proof; [authoring](authoring.md) covers the
89
+ Scenario quality is behavioral proof; [authoring](authoring.md) covers the
90
90
  judgment layer lint and evals cannot grade.
package/docs/usage.md CHANGED
@@ -1,60 +1,62 @@
1
- # usage
1
+ # Usage
2
2
 
3
- every subcommand resolves one root, `--root <dir>` or the current directory.
3
+ Every subcommand resolves one root, `--root <dir>` or the current directory.
4
4
  `lint` also takes the root as a positional, because that is the shape CI reaches
5
5
  for first.
6
6
 
7
- ## lint
7
+ ## Lint
8
8
 
9
9
  ```sh
10
10
  skillcheck lint # lints the current repo
11
11
  skillcheck lint ../other # lints another root
12
12
  ```
13
13
 
14
- checks each `<root>/skills/<skill>/`:
14
+ Checks each `<root>/skills/<skill>/`:
15
15
 
16
- - frontmatter opens with `---` on line 1 and closes
17
- - keys are `name`, `description`, `disable-model-invocation` and nothing else,
16
+ - Frontmatter opens with `---` on line 1 and closes
17
+ - Keys are `name`, `description`, `disable-model-invocation` and nothing else,
18
18
  each at most once
19
19
  - `name` equals the directory name; `description` is non-empty
20
20
  - `disable-model-invocation`, when present, is the bare YAML boolean `true`.
21
- a quoted `"true"` is an error
22
- - relative links in the body resolve on disk
21
+ A quoted `"true"` is an error
22
+ - Relative links in the body resolve on disk
23
23
 
24
- code spans and fenced blocks are stripped before links are checked, so example
25
- links never fail. external schemes and `#anchors` pass. dot-directories under
24
+ Code spans and fenced blocks are stripped before links are checked, so example
25
+ links never fail. External schemes and `#anchors` pass. Dot-directories under
26
26
  `skills/` (`.claude-plugin`) are plugin metadata, not packages, and are skipped.
27
27
 
28
- findings print one per line, relative to the linted root, then a count. exit 0
28
+ Findings print one per line, relative to the linted root, then a count. Exit 0
29
29
  clean, 1 with findings.
30
30
 
31
- ## run
31
+ ## Run
32
32
 
33
33
  ```sh
34
34
  skillcheck run skills/<skill>/evals/<scenario>
35
35
  skillcheck run <scenario-dir> --agent MODEL --judge MODEL --harness codex --max-turns 80
36
36
  ```
37
37
 
38
- materializes the scenario into `<root>/.skillcheck/scratch/<name>/workdir`,
38
+ Materializes the scenario into `<root>/.skillcheck/scratch/<name>/workdir`,
39
39
  installs the skill under test into that workdir, drives the agent, and grades
40
- the files it wrote. exit 0 pass, 1 graded fail, 2 error (promptfoo produced no
41
- usable result).
40
+ the files it wrote. Exit 0 pass, 1 graded fail, 2 error (promptfoo produced no
41
+ usable result, or the optional eval peers are not installed — the message
42
+ carries the exact `pnpm add` command; see
43
+ [adoption](adoption.md#evals)).
42
44
 
43
- a test that errored was never graded, so it exits 2, prints the provider's
44
- message, and writes no provenance sidecar. it is never reported as
45
+ A test that errored was never graded, so it exits 2, prints the provider's
46
+ message, and writes no provenance sidecar. It is never reported as
45
47
  `FAIL score=0.0000`; only a real judged verdict can fail a run.
46
48
 
47
- defaults: `--harness claude`, agent `claude-opus-5`, judge `claude-opus-5`,
48
- `--max-turns 50`. on the codex and cursor harnesses, omitting `--agent` leaves
49
+ Defaults: `--harness claude`, agent `claude-opus-5`, judge `claude-opus-5`,
50
+ `--max-turns 50`. On the codex and cursor harnesses, omitting `--agent` leaves
49
51
  the model to that CLI's own default.
50
52
 
51
53
  `--harness cursor` drives the scenario through the Cursor Agent CLI
52
54
  (`cursor-agent` on PATH) with the skill installed under `.cursor/skills/`;
53
- `--agent` names a Cursor model id, e.g. `composer-2.5`. there is no promptfoo
55
+ `--agent` names a Cursor model id, e.g. `composer-2.5`. There is no promptfoo
54
56
  cursor provider, so the run uses this package's own provider module, which
55
57
  replays the CLI's `stream-json` output: the `result` event becomes the graded
56
58
  output and `SKILL.md` reads under `.cursor/skills/` become the `skill-used`
57
- evidence. the judge leg is unchanged.
59
+ evidence. The judge leg is unchanged.
58
60
 
59
61
  `--judge` takes either a bare Claude model (graded through the Anthropic
60
62
  selection in [auth](#auth)) or a provider-qualified promptfoo id, passed
@@ -64,55 +66,55 @@ through verbatim:
64
66
  skillcheck run <scenario-dir> --judge openai:chat:gpt-5.6-sol --judge-effort high
65
67
  ```
66
68
 
67
- a provider-qualified judge authenticates through that provider's own env
69
+ A provider-qualified judge authenticates through that provider's own env
68
70
  (`OPENAI_API_KEY`, plus `OPENAI_BASE_URL` for a gateway) and is recorded
69
71
  verbatim in the scorecard's `judge_model` column. `--judge-effort`
70
72
  (minimal|low|medium|high) sets `reasoning_effort` and requires a
71
73
  provider-qualified judge; the Anthropic judge does not take one.
72
74
 
73
- ## sweep
75
+ ## Sweep
74
76
 
75
77
  ```sh
76
78
  skillcheck sweep # only scenarios without results
77
79
  skillcheck sweep --all # rerun everything
78
80
  ```
79
81
 
80
- walks `<root>/skills/*/evals/*` and `<root>/cli/*/skills/*/evals/*`, in sorted
81
- order, sequentially. a scenario needs both `task.md` and `criteria.json` to be
82
- discovered. exit 2 if anything errored, 1 if anything failed, else 0.
82
+ Walks `<root>/skills/*/evals/*` and `<root>/cli/*/skills/*/evals/*`, in sorted
83
+ order, sequentially. A scenario needs both `task.md` and `criteria.json` to be
84
+ discovered. Exit 2 if anything errored, 1 if anything failed, else 0.
83
85
 
84
- `EVALS_CONCURRENCY` is passed to promptfoo as `-j` (default 4). it parallelizes
86
+ `EVALS_CONCURRENCY` is passed to promptfoo as `-j` (default 4). It parallelizes
85
87
  within one scenario, not across them.
86
88
 
87
- one known failure mode: judge calls through a gateway can drop at the transport
89
+ One known failure mode: judge calls through a gateway can drop at the transport
88
90
  layer ([uinaf/agent-platform#28](https://github.com/uinaf/agent-platform/issues/28)).
89
- that surfaces as an ERROR with no usable result, not as a graded FAIL, and the
91
+ That surfaces as an ERROR with no usable result, not as a graded FAIL, and the
90
92
  mitigation is a rerun. `sweep` without `--all` resumes, so a rerun only picks up
91
93
  what is missing.
92
94
 
93
- ## summarize
95
+ ## Summarize
94
96
 
95
97
  ```sh
96
98
  skillcheck summarize [--allow-mixed]
97
99
  ```
98
100
 
99
- reduces `<root>/.skillcheck/results/*.json` into
101
+ Reduces `<root>/.skillcheck/results/*.json` into
100
102
  `<root>/.skillcheck/scorecards/<UTC-date>.json`: one entry per scenario with
101
103
  skill, scenario, harness, tree sha, score, pass, both models, latency, tokens.
102
104
 
103
- if a scorecard for today already exists, the two are merged on
105
+ If a scorecard for today already exists, the two are merged on
104
106
  `(skill, scenario, harness)`: entries from this run win, entries it did not
105
- touch survive, and the merge is reported on stdout. summarizing after rerunning
107
+ touch survive, and the merge is reported on stdout. Summarizing after rerunning
106
108
  six of twenty-nine scenarios therefore leaves twenty-nine rows in the file, not
107
- six. a same-date file that cannot be parsed stops the write instead of being
109
+ six. A same-date file that cannot be parsed stops the write instead of being
108
110
  overwritten.
109
111
 
110
- files that are not promptfoo results are skipped with a warning rather than
112
+ Files that are not promptfoo results are skipped with a warning rather than
111
113
  failing the reduction.
112
114
 
113
- ## provenance
115
+ ## Provenance
114
116
 
115
- each successful run writes a `<name>.meta.json` sidecar next to its result:
117
+ Each successful run writes a `<name>.meta.json` sidecar next to its result:
116
118
 
117
119
  ```json
118
120
  {
@@ -125,26 +127,26 @@ each successful run writes a `<name>.meta.json` sidecar next to its result:
125
127
 
126
128
  `summarize` reads those sidecars and refuses to mix skills-tree revisions in one
127
129
  scorecard unless `--allow-mixed`, in which case the top-level `skills_tree_sha`
128
- becomes `mixed` and per-entry shas remain. a result with no sidecar reduces as
130
+ becomes `mixed` and per-entry shas remain. A result with no sidecar reduces as
129
131
  `unattested`.
130
132
 
131
- ## state
133
+ ## State
132
134
 
133
135
  `<root>/.skillcheck/` holds `scratch/` and `results/`, both disposable and safe
134
- to gitignore, and `scorecards/`, which is meant to be committed. nothing is ever
136
+ to gitignore, and `scorecards/`, which is meant to be committed. Nothing is ever
135
137
  written inside the installed package.
136
138
 
137
- ## auth
139
+ ## Auth
138
140
 
139
- | variable | effect |
141
+ | Variable | Effect |
140
142
  | --------------------------------------------- | ----------------------------------------------------------------- |
141
- | `ANTHROPIC_BASE_URL` + `ANTHROPIC_AUTH_TOKEN` | claude agent and judge go through a gateway |
142
- | none of the above | falls back to the local Claude Code session |
143
- | `ANTHROPIC_API_KEY` | judge grades over `anthropic:messages:<model>` instead of the SDK |
144
- | `CODEX_HOME` (default `~/.codex`) | where the codex harness finds the local `codex` CLI login |
145
- | `OPENAI_API_KEY` | codex agent auth when there is no local login |
146
- | `CURSOR_API_KEY` | cursor agent auth; a logged-in `cursor-agent` also works |
147
- | `OPENAI_API_KEY` + `OPENAI_BASE_URL` | a provider-qualified `--judge openai:…`, optionally via a gateway |
148
-
149
- a bare `--judge` model stays on the Anthropic selection regardless of the
143
+ | `ANTHROPIC_BASE_URL` + `ANTHROPIC_AUTH_TOKEN` | The claude agent and judge go through a gateway |
144
+ | None of the above | Falls back to the local Claude Code session |
145
+ | `ANTHROPIC_API_KEY` | Judge grades over `anthropic:messages:<model>` instead of the SDK |
146
+ | `CODEX_HOME` (default `~/.codex`) | Where the codex harness finds the local `codex` CLI login |
147
+ | `OPENAI_API_KEY` | Agent auth for codex when there is no local login |
148
+ | `CURSOR_API_KEY` | Agent auth for cursor; a logged-in `cursor-agent` also works |
149
+ | `OPENAI_API_KEY` + `OPENAI_BASE_URL` | A provider-qualified `--judge openai:…`, optionally via a gateway |
150
+
151
+ A bare `--judge` model stays on the Anthropic selection regardless of the
150
152
  agent harness; a provider-qualified `--judge` uses that provider's env instead.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "0.4.0",
3
+ "version": "0.5.0",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {
@@ -31,16 +31,30 @@
31
31
  "prepare": "vp config --no-agent",
32
32
  "prepublishOnly": "pnpm run verify"
33
33
  },
34
- "dependencies": {
34
+ "devDependencies": {
35
35
  "@anthropic-ai/claude-agent-sdk": "^0.3.233",
36
36
  "@openai/codex-sdk": "^0.147.0",
37
- "promptfoo": "^0.122.0"
38
- },
39
- "devDependencies": {
40
37
  "@types/node": "^26.2.0",
38
+ "promptfoo": "^0.122.0",
41
39
  "vite": "catalog:",
42
40
  "vite-plus": "catalog:"
43
41
  },
42
+ "peerDependencies": {
43
+ "@anthropic-ai/claude-agent-sdk": "^0.3.233",
44
+ "@openai/codex-sdk": "^0.147.0",
45
+ "promptfoo": "^0.122.0"
46
+ },
47
+ "peerDependenciesMeta": {
48
+ "@anthropic-ai/claude-agent-sdk": {
49
+ "optional": true
50
+ },
51
+ "@openai/codex-sdk": {
52
+ "optional": true
53
+ },
54
+ "promptfoo": {
55
+ "optional": true
56
+ }
57
+ },
44
58
  "engines": {
45
59
  "node": ">=24"
46
60
  },