@uinaf/skillcheck 0.3.1 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -22
- package/dist/cli.js +48 -8
- package/dist/scenario.js +23 -3
- package/docs/adoption.md +29 -19
- package/docs/authoring.md +34 -34
- package/docs/releasing.md +24 -24
- package/docs/scenarios.md +31 -31
- package/docs/usage.md +67 -49
- package/package.json +19 -5
package/README.md
CHANGED
|
@@ -2,25 +2,28 @@
|
|
|
2
2
|
|
|
3
3
|
# uinaf/skillcheck
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
Lint and eval harness for agent skills. One CLI with two halves: a keyless
|
|
6
6
|
structural lint any repo can run in CI, and a promptfoo-driven eval loop that
|
|
7
7
|
grades what a skill actually makes an agent do.
|
|
8
8
|
|
|
9
|
-
|
|
10
|
-
uinaf-specific.
|
|
9
|
+
Built for [uinaf](https://uinaf.dev) skill repos. Nothing in it is
|
|
10
|
+
uinaf-specific. It ships no opinion about what a skill should say, only about
|
|
11
11
|
where skills sit and how a scenario is scored.
|
|
12
12
|
|
|
13
|
-
##
|
|
13
|
+
## Install
|
|
14
14
|
|
|
15
15
|
```sh
|
|
16
16
|
pnpm add -D @uinaf/skillcheck
|
|
17
17
|
```
|
|
18
18
|
|
|
19
|
-
|
|
20
|
-
so a runner using `--ignore-scripts` is fine
|
|
21
|
-
|
|
19
|
+
Node 24 or newer. The package ships compiled ESM, runs no install scripts, and
|
|
20
|
+
has no regular dependencies, so a runner using `--ignore-scripts` is fine and
|
|
21
|
+
the lint-only install stays at a handful of packages. The promptfoo eval engine
|
|
22
|
+
and provider SDKs are optional peers, installed only on the operator machine
|
|
23
|
+
that runs evals ([adoption](docs/adoption.md)). Consumers still pinned to the
|
|
24
|
+
pre-npm git tags are covered there too.
|
|
22
25
|
|
|
23
|
-
##
|
|
26
|
+
## Use
|
|
24
27
|
|
|
25
28
|
```sh
|
|
26
29
|
skillcheck lint # structural lint, no credentials
|
|
@@ -29,12 +32,12 @@ skillcheck sweep && skillcheck summarize # every scenario, then a scorecard
|
|
|
29
32
|
```
|
|
30
33
|
|
|
31
34
|
`lint` needs nothing. `run` and `sweep` need model auth, which is why sweeps
|
|
32
|
-
stay operator-run and consumer repos never hold credentials.
|
|
35
|
+
stay operator-run and consumer repos never hold credentials. Every command,
|
|
33
36
|
flag, and auth variable is in [usage](docs/usage.md).
|
|
34
37
|
|
|
35
|
-
##
|
|
38
|
+
## Layout contract
|
|
36
39
|
|
|
37
|
-
|
|
40
|
+
Frozen, not configurable. Every command reads one root: `--root <dir>`, or the
|
|
38
41
|
current directory.
|
|
39
42
|
|
|
40
43
|
```text
|
|
@@ -47,18 +50,18 @@ current directory.
|
|
|
47
50
|
`cli/*/skills/<skill>/` is scanned too, for repos that keep a skill next to the
|
|
48
51
|
CLI it documents.
|
|
49
52
|
|
|
50
|
-
##
|
|
53
|
+
## Docs
|
|
51
54
|
|
|
52
|
-
|
|
|
55
|
+
| Doc | When |
|
|
53
56
|
| --------------------------------------------------------------- | ------------------------------------- |
|
|
54
|
-
| [
|
|
55
|
-
| [
|
|
56
|
-
| [
|
|
57
|
-
| [
|
|
58
|
-
| [
|
|
59
|
-
| [
|
|
60
|
-
| [
|
|
61
|
-
|
|
62
|
-
##
|
|
57
|
+
| [Usage](docs/usage.md) | Every subcommand, flag, and auth path |
|
|
58
|
+
| [Scenarios](docs/scenarios.md) | Writing an eval scenario |
|
|
59
|
+
| [Authoring](docs/authoring.md) | Writing and auditing the skill itself |
|
|
60
|
+
| [Adoption](docs/adoption.md) | Wiring the lint into another repo |
|
|
61
|
+
| [Releasing](docs/releasing.md) | The npm pipeline |
|
|
62
|
+
| [Contributing](CONTRIBUTING.md) | Local setup and the verify gate |
|
|
63
|
+
| [Security](https://github.com/uinaf/skillcheck/security/policy) | Reporting a vulnerability |
|
|
64
|
+
|
|
65
|
+
## License
|
|
63
66
|
|
|
64
67
|
MIT · undefined is not a function LLC
|
package/dist/cli.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import { lintSkills } from "./lint.js";
|
|
3
|
-
import { generateRun, runNameFor } from "./scenario.js";
|
|
3
|
+
import { generateRun, requiredEvalPackages, resolvePackageDir, runNameFor } from "./scenario.js";
|
|
4
4
|
import { execFileSync, spawnSync } from "node:child_process";
|
|
5
5
|
import fs from "node:fs";
|
|
6
6
|
import path from "node:path";
|
|
@@ -28,6 +28,7 @@ function parseArgs(argv) {
|
|
|
28
28
|
"--root",
|
|
29
29
|
"--agent",
|
|
30
30
|
"--judge",
|
|
31
|
+
"--judge-effort",
|
|
31
32
|
"--harness",
|
|
32
33
|
"--max-turns"
|
|
33
34
|
]);
|
|
@@ -60,13 +61,49 @@ function stateDirs(root) {
|
|
|
60
61
|
function runOptions(flags) {
|
|
61
62
|
const harness = flags.get("--harness") ?? "claude";
|
|
62
63
|
if (harness !== "claude" && harness !== "codex" && harness !== "cursor") fail(`--harness must be claude, codex, or cursor, got ${harness}`);
|
|
64
|
+
const agent = flags.get("--agent");
|
|
65
|
+
const judgeModel = flags.get("--judge") ?? "claude-opus-5";
|
|
66
|
+
const judgeEffort = flags.get("--judge-effort");
|
|
67
|
+
if (judgeEffort !== void 0) {
|
|
68
|
+
if (![
|
|
69
|
+
"minimal",
|
|
70
|
+
"low",
|
|
71
|
+
"medium",
|
|
72
|
+
"high"
|
|
73
|
+
].includes(judgeEffort)) fail(`--judge-effort must be minimal, low, medium, or high, got ${judgeEffort}`);
|
|
74
|
+
if (!judgeModel.includes(":")) fail("--judge-effort needs a provider-qualified --judge (e.g. openai:chat:gpt-5.6-sol); the Anthropic judge does not take a reasoning effort");
|
|
75
|
+
}
|
|
63
76
|
return {
|
|
64
77
|
harness,
|
|
65
|
-
agentModel:
|
|
66
|
-
judgeModel
|
|
78
|
+
agentModel: agent,
|
|
79
|
+
judgeModel,
|
|
80
|
+
judgeEffort,
|
|
67
81
|
maxTurns: flags.has("--max-turns") ? parseMaxTurns(flags.get("--max-turns")) : void 0
|
|
68
82
|
};
|
|
69
83
|
}
|
|
84
|
+
function ensureEvalPackages(opts) {
|
|
85
|
+
const missing = requiredEvalPackages(opts, process.env.ANTHROPIC_API_KEY !== void 0).filter((pkg) => resolvePackageDir(pkg) === void 0);
|
|
86
|
+
if (missing.length === 0) return;
|
|
87
|
+
const peers = JSON.parse(fs.readFileSync(path.join(packageDir, "package.json"), "utf8")).peerDependencies ?? {};
|
|
88
|
+
const specs = missing.map((pkg) => `"${pkg}@${peers[pkg] ?? "latest"}"`).join(" ");
|
|
89
|
+
console.error([
|
|
90
|
+
`missing eval package(s): ${missing.join(", ")}`,
|
|
91
|
+
"",
|
|
92
|
+
"The eval engine is an optional peer so `skillcheck lint` installs stay",
|
|
93
|
+
"small. Evals are operator-run; install the peers next to @uinaf/skillcheck:",
|
|
94
|
+
"",
|
|
95
|
+
` pnpm add -D ${specs}`
|
|
96
|
+
].join("\n"));
|
|
97
|
+
process.exit(2);
|
|
98
|
+
}
|
|
99
|
+
function promptfooEntry() {
|
|
100
|
+
const dir = resolvePackageDir("promptfoo");
|
|
101
|
+
if (dir === void 0) throw new Error("promptfoo is not installed");
|
|
102
|
+
const bin = JSON.parse(fs.readFileSync(path.join(dir, "package.json"), "utf8")).bin;
|
|
103
|
+
const rel = typeof bin === "string" ? bin : bin?.promptfoo;
|
|
104
|
+
if (typeof rel !== "string") throw new Error(`promptfoo at ${dir} declares no bin`);
|
|
105
|
+
return path.join(dir, rel);
|
|
106
|
+
}
|
|
70
107
|
function classifyResult(raw) {
|
|
71
108
|
const root = raw;
|
|
72
109
|
const res = root?.results?.results?.[0];
|
|
@@ -103,8 +140,8 @@ function runScenario(scenarioDir, opts, root) {
|
|
|
103
140
|
fs.rmSync(resultPath, { force: true });
|
|
104
141
|
fs.rmSync(metaPath(resultPath), { force: true });
|
|
105
142
|
const sha = gitHead(root);
|
|
106
|
-
const rc = spawnSync(
|
|
107
|
-
|
|
143
|
+
const rc = spawnSync(process.execPath, [
|
|
144
|
+
promptfooEntry(),
|
|
108
145
|
"eval",
|
|
109
146
|
"--no-cache",
|
|
110
147
|
"--no-progress-bar",
|
|
@@ -167,8 +204,10 @@ function discoverScenarios(root) {
|
|
|
167
204
|
}
|
|
168
205
|
function cmdRun(argv) {
|
|
169
206
|
const { positional, flags } = parseArgs(argv);
|
|
170
|
-
if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--harness claude|codex|cursor]");
|
|
171
|
-
const
|
|
207
|
+
if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--judge-effort EFFORT] [--harness claude|codex|cursor]");
|
|
208
|
+
const opts = runOptions(flags);
|
|
209
|
+
ensureEvalPackages(opts);
|
|
210
|
+
const o = runScenario(positional[0], opts, resolveRoot(flags));
|
|
172
211
|
if (o.score === void 0) {
|
|
173
212
|
console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
|
|
174
213
|
process.exit(2);
|
|
@@ -181,6 +220,7 @@ function cmdSweep(argv) {
|
|
|
181
220
|
if (positional.length > 0) fail("usage: skillcheck sweep [--root DIR] [--all]");
|
|
182
221
|
const root = resolveRoot(flags);
|
|
183
222
|
const opts = runOptions(flags);
|
|
223
|
+
ensureEvalPackages(opts);
|
|
184
224
|
const all = flags.get("--all") === true;
|
|
185
225
|
const resultsDir = stateDirs(root).results;
|
|
186
226
|
let passed = 0, failed = 0, errored = 0, skipped = 0;
|
|
@@ -243,7 +283,7 @@ function reduceResults(dir, allowMixed) {
|
|
|
243
283
|
score: res.score,
|
|
244
284
|
pass: res.success,
|
|
245
285
|
agent_model: provider?.config?.model ?? `${harness}-default`,
|
|
246
|
-
judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? "unknown",
|
|
286
|
+
judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? judge?.id ?? "unknown",
|
|
247
287
|
latency_ms: res.latencyMs,
|
|
248
288
|
tokens: (res.tokenUsage?.total ?? 0) + (res.tokenUsage?.assertions?.total ?? 0)
|
|
249
289
|
});
|
package/dist/scenario.js
CHANGED
|
@@ -155,7 +155,10 @@ function buildConfig(s, workdir, manifestPath, opts, paths) {
|
|
|
155
155
|
prompts: ["{{task}}"],
|
|
156
156
|
providers: [agentProvider(opts, workdir, s.skill, paths)],
|
|
157
157
|
defaultTest: { options: {
|
|
158
|
-
provider:
|
|
158
|
+
provider: opts.judgeModel.includes(":") ? opts.judgeEffort === void 0 ? opts.judgeModel : {
|
|
159
|
+
id: opts.judgeModel,
|
|
160
|
+
config: { reasoning_effort: opts.judgeEffort }
|
|
161
|
+
} : process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
|
|
159
162
|
id: "anthropic:claude-agent-sdk",
|
|
160
163
|
config: {
|
|
161
164
|
model: opts.judgeModel,
|
|
@@ -210,8 +213,25 @@ function buildConfig(s, workdir, manifestPath, opts, paths) {
|
|
|
210
213
|
}
|
|
211
214
|
const SDK_PACKAGES = ["@anthropic-ai/claude-agent-sdk", "@openai/codex-sdk"];
|
|
212
215
|
function sdkNodeModulesDir() {
|
|
216
|
+
for (const pkg of SDK_PACKAGES) {
|
|
217
|
+
const dir = holdingNodeModules(pkg);
|
|
218
|
+
if (dir !== void 0) return dir;
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
function holdingNodeModules(pkg) {
|
|
213
222
|
const require = createRequire(import.meta.url);
|
|
214
|
-
for (const
|
|
223
|
+
for (const dir of require.resolve.paths(pkg) ?? []) if (fs.existsSync(path.join(dir, pkg, "package.json"))) return dir;
|
|
224
|
+
}
|
|
225
|
+
function resolvePackageDir(pkg) {
|
|
226
|
+
const dir = holdingNodeModules(pkg);
|
|
227
|
+
return dir === void 0 ? void 0 : path.join(dir, pkg);
|
|
228
|
+
}
|
|
229
|
+
function requiredEvalPackages(opts, hasAnthropicKey) {
|
|
230
|
+
const pkgs = ["promptfoo"];
|
|
231
|
+
const sdkJudge = !opts.judgeModel.includes(":") && !hasAnthropicKey;
|
|
232
|
+
if (opts.harness === "claude" || sdkJudge) pkgs.push("@anthropic-ai/claude-agent-sdk");
|
|
233
|
+
if (opts.harness === "codex") pkgs.push("@openai/codex-sdk");
|
|
234
|
+
return pkgs;
|
|
215
235
|
}
|
|
216
236
|
function generateRun(scenarioDir, opts, paths) {
|
|
217
237
|
const s = loadScenario(scenarioDir);
|
|
@@ -236,4 +256,4 @@ function generateRun(scenarioDir, opts, paths) {
|
|
|
236
256
|
};
|
|
237
257
|
}
|
|
238
258
|
//#endregion
|
|
239
|
-
export { buildConfig, generateRun, isHiddenSkill, loadScenario, materialize, runNameFor, sdkNodeModulesDir, stripHiddenFlag };
|
|
259
|
+
export { buildConfig, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag };
|
package/docs/adoption.md
CHANGED
|
@@ -1,15 +1,15 @@
|
|
|
1
|
-
#
|
|
1
|
+
# Adopting it in a repo
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
model auth.
|
|
3
|
+
Two separate decisions: run the lint in CI, and run evals on a machine that has
|
|
4
|
+
model auth. Only the first belongs in a consumer repo.
|
|
5
5
|
|
|
6
|
-
##
|
|
6
|
+
## Lint in CI
|
|
7
7
|
|
|
8
8
|
```sh
|
|
9
9
|
pnpm add -D @uinaf/skillcheck
|
|
10
10
|
```
|
|
11
11
|
|
|
12
|
-
|
|
12
|
+
Runners that install with `--ignore-scripts` are fine: the package ships
|
|
13
13
|
compiled ESM and has no install, prepare, or postinstall script.
|
|
14
14
|
|
|
15
15
|
```json
|
|
@@ -32,49 +32,59 @@ jobs:
|
|
|
32
32
|
- run: pnpm run skills:lint
|
|
33
33
|
```
|
|
34
34
|
|
|
35
|
-
|
|
35
|
+
The job needs no secrets and no network beyond the install. Node 24 is the
|
|
36
36
|
floor.
|
|
37
37
|
|
|
38
|
-
|
|
38
|
+
Run it through the script rather than a bare `npx skillcheck`: the script
|
|
39
39
|
resolves the version the repo pinned, and `npx` would resolve the latest one on
|
|
40
40
|
the registry.
|
|
41
41
|
|
|
42
|
-
##
|
|
42
|
+
## Evals
|
|
43
43
|
|
|
44
|
-
|
|
45
|
-
operator machine or a job that already holds gateway auth
|
|
44
|
+
Sweeps need model credentials, so they stay off consumer CI and run from an
|
|
45
|
+
operator machine or a job that already holds gateway auth. They also need the
|
|
46
|
+
eval engine, which is an optional peer precisely so the lint-only install
|
|
47
|
+
above stays small — install it next to the package on the operator machine:
|
|
48
|
+
|
|
49
|
+
```sh
|
|
50
|
+
pnpm add -D promptfoo @anthropic-ai/claude-agent-sdk @openai/codex-sdk
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
`run` and `sweep` check for the peers the selected harness needs and exit 2
|
|
54
|
+
with that install command when they are missing, so a lint-only install never
|
|
55
|
+
crashes into a resolution error. Then:
|
|
46
56
|
|
|
47
57
|
```sh
|
|
48
58
|
skillcheck sweep # resumes: only scenarios without results
|
|
49
59
|
skillcheck summarize # writes .skillcheck/scorecards/<UTC-date>.json
|
|
50
60
|
```
|
|
51
61
|
|
|
52
|
-
|
|
62
|
+
Commit `.skillcheck/scorecards/`. Gitignore the rest:
|
|
53
63
|
|
|
54
64
|
```gitignore
|
|
55
65
|
.skillcheck/results/
|
|
56
66
|
.skillcheck/scratch/
|
|
57
67
|
```
|
|
58
68
|
|
|
59
|
-
|
|
69
|
+
A scorecard is only comparable against the tree it graded, which is why every
|
|
60
70
|
result carries the root repo's HEAD and `summarize` refuses to mix revisions
|
|
61
71
|
without `--allow-mixed`.
|
|
62
72
|
|
|
63
|
-
##
|
|
73
|
+
## Upgrading
|
|
64
74
|
|
|
65
|
-
|
|
75
|
+
Bump the version in `package.json` and rerun the sweep. Results carry the
|
|
66
76
|
`tool_version` that produced them, so a scorecard says which harness build it
|
|
67
77
|
came from as well as which skills tree.
|
|
68
78
|
|
|
69
|
-
##
|
|
79
|
+
## The older install path
|
|
70
80
|
|
|
71
|
-
|
|
81
|
+
Before the package was published, consumers installed it from a git tag:
|
|
72
82
|
|
|
73
83
|
```sh
|
|
74
84
|
npm i -D github:uinaf/skillcheck#v0.1.3
|
|
75
85
|
```
|
|
76
86
|
|
|
77
|
-
|
|
78
|
-
12 consumers needed `allow-git=root` in `.npmrc` to accept the spec at all.
|
|
79
|
-
registry install needs none of that.
|
|
87
|
+
Those tags are frozen and still work: they carry a committed `dist/`, and npm
|
|
88
|
+
12 consumers needed `allow-git=root` in `.npmrc` to accept the spec at all. The
|
|
89
|
+
registry install needs none of that. Tags from `v0.1.4` on are npm releases and
|
|
80
90
|
carry no `dist/`, so a git spec pointing at one will not run.
|
package/docs/authoring.md
CHANGED
|
@@ -1,70 +1,70 @@
|
|
|
1
|
-
#
|
|
1
|
+
# Authoring and auditing skills
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
cheap to load.
|
|
3
|
+
What `lint` and evals cannot judge: whether a skill is worth routing to and
|
|
4
|
+
cheap to load. Use this when writing a skill or auditing one. Evidence beats
|
|
5
5
|
stylistic preference; run `skillcheck lint` first and let this cover the rest.
|
|
6
6
|
|
|
7
|
-
##
|
|
7
|
+
## Metadata and discovery
|
|
8
8
|
|
|
9
9
|
- `name` is concrete and easy to say out loud. `helper`, `tools`, `utils` are
|
|
10
10
|
discovery smells.
|
|
11
11
|
- `description` is third person and says both what the skill does and when to
|
|
12
|
-
use it.
|
|
12
|
+
use it. It is an always-loaded retrieval pointer: front-load the concrete
|
|
13
13
|
action or domain that should activate it.
|
|
14
|
-
-
|
|
14
|
+
- One trigger per materially distinct request branch. Collapse synonyms that
|
|
15
15
|
rename the same branch.
|
|
16
|
-
-
|
|
16
|
+
- State the main overlap boundary without naming another skill.
|
|
17
17
|
|
|
18
|
-
##
|
|
18
|
+
## Body shape
|
|
19
19
|
|
|
20
|
-
-
|
|
20
|
+
- Keep `SKILL.md` on workflow, principles, boundaries, and routing. Lead with
|
|
21
21
|
the task, not a bibliography.
|
|
22
|
-
-
|
|
22
|
+
- Assume the model is smart; spend tokens on repo-specific judgment. Delete any
|
|
23
23
|
instruction that would not change a capable model's behavior.
|
|
24
|
-
-
|
|
24
|
+
- Match freedom to risk: high for contextual judgment, medium when a preferred
|
|
25
25
|
pattern exists, low for fragile operations.
|
|
26
|
-
-
|
|
26
|
+
- Say what evidence to gather and what a complete result includes. End each
|
|
27
27
|
step with an observable completion condition, not "understood" or "handled".
|
|
28
28
|
|
|
29
|
-
##
|
|
29
|
+
## Progressive disclosure
|
|
30
30
|
|
|
31
|
-
-
|
|
32
|
-
`SKILL.md`, each with a task-shaped retrieval job.
|
|
31
|
+
- Durable detail, rubrics, and long examples go in `references/`, one hop from
|
|
32
|
+
`SKILL.md`, each with a task-shaped retrieval job. Material every path needs
|
|
33
33
|
stays inline.
|
|
34
|
-
-
|
|
34
|
+
- For repeated deterministic work, route to the target's existing framework,
|
|
35
35
|
schema, task graph, or library; otherwise add a tested module in the
|
|
36
36
|
project's primary language, not ad-hoc shell rendered as prose.
|
|
37
|
-
-
|
|
37
|
+
- When executable code belongs to another maintained project, link the exact
|
|
38
38
|
public artifact and state the contract it demonstrates; do not fork it into
|
|
39
39
|
prose.
|
|
40
|
-
-
|
|
41
|
-
next steps locally.
|
|
40
|
+
- A package stays independently usable: state prerequisites and out-of-scope
|
|
41
|
+
next steps locally. Never invoke, import, or assume a sibling skill.
|
|
42
42
|
|
|
43
|
-
##
|
|
43
|
+
## Audit
|
|
44
44
|
|
|
45
|
-
|
|
45
|
+
Grade each dimension strong, mixed, or weak:
|
|
46
46
|
|
|
47
|
-
|
|
|
47
|
+
| Dimension | Question |
|
|
48
48
|
| ---------------------- | --------------------------------------------------------------------- |
|
|
49
|
-
|
|
|
50
|
-
|
|
|
51
|
-
|
|
|
52
|
-
|
|
|
53
|
-
|
|
|
54
|
-
|
|
|
49
|
+
| Discovery | Does metadata alone route a realistic request here |
|
|
50
|
+
| Workflow | Does the body say how to begin, what evidence to gather, when to stop |
|
|
51
|
+
| Progressive disclosure | Is detail in the right file |
|
|
52
|
+
| Repo fit | Are links, commands, and conventions current |
|
|
53
|
+
| Verification | Is the strongest mechanical check named, plus a real evidence loop |
|
|
54
|
+
| Boundaries | Are limits and next steps stated without leaning on a sibling skill |
|
|
55
55
|
|
|
56
|
-
|
|
56
|
+
Blockers, must-fix: invalid frontmatter; a description that fails discovery;
|
|
57
57
|
stale commands, paths, or links; a workflow with no start, evidence loop, or
|
|
58
58
|
completion; conflicts with the repo's guidance; sibling-skill dependencies.
|
|
59
59
|
|
|
60
|
-
|
|
60
|
+
Major findings: vague name; synonym-stuffed description; bloated `SKILL.md`;
|
|
61
61
|
missing or muddy boundaries; prose re-inventing a deterministic tool; abstract
|
|
62
62
|
examples.
|
|
63
63
|
|
|
64
|
-
##
|
|
64
|
+
## Improve
|
|
65
65
|
|
|
66
|
-
|
|
67
|
-
change that improves activation, decision quality, or proof.
|
|
66
|
+
Fix blockers first, then the highest-leverage majors. Prefer the smallest
|
|
67
|
+
change that improves activation, decision quality, or proof. When pruning,
|
|
68
68
|
measure common-path context for representative requests; line count alone does
|
|
69
|
-
not reveal retrieval cost.
|
|
69
|
+
not reveal retrieval cost. After edits, rerun `skillcheck lint` and the repo's
|
|
70
70
|
gate, and rerun evals when behavior was the thing changed.
|
package/docs/releasing.md
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
|
-
#
|
|
1
|
+
# Releasing
|
|
2
2
|
|
|
3
|
-
##
|
|
3
|
+
## Pipeline
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
A push to `main` runs one workflow, `.github/workflows/release.yml`:
|
|
6
6
|
|
|
7
7
|
```text
|
|
8
8
|
verify ──┐
|
|
@@ -12,69 +12,69 @@ scan ────┘
|
|
|
12
12
|
|
|
13
13
|
`verify` and `scan` are the shared gate: `verify` is called from `verify.yml`,
|
|
14
14
|
and `scan` calls the shared scan in `uinaf/.github`, the same one `scan.yml`
|
|
15
|
-
runs for pull requests.
|
|
15
|
+
runs for pull requests. Keep it that way: a second copy of the gate on a push-to-`main` workflow
|
|
16
16
|
races this one over the same commit.
|
|
17
17
|
|
|
18
|
-
|
|
18
|
+
The file name `release.yml` is load-bearing. See below.
|
|
19
19
|
|
|
20
20
|
## npm
|
|
21
21
|
|
|
22
22
|
`@uinaf/skillcheck` publishes from `.github/workflows/release.yml` via npm
|
|
23
|
-
Trusted Publishing (OIDC) and the `uinaf-releaser` GitHub App.
|
|
23
|
+
Trusted Publishing (OIDC) and the `uinaf-releaser` GitHub App. There is no npm
|
|
24
24
|
token in this repository, in its environments, or in the organization.
|
|
25
25
|
|
|
26
|
-
|
|
26
|
+
Required on the `release` GitHub Environment:
|
|
27
27
|
|
|
28
|
-
|
|
|
28
|
+
| Name | Kind | Purpose |
|
|
29
29
|
| ------------------------------- | ------ | ------------------------------------------- |
|
|
30
30
|
| `UINAF_RELEASE_APP_CLIENT_ID` | var | GitHub App client id for the releaser bot |
|
|
31
31
|
| `UINAF_RELEASE_APP_PRIVATE_KEY` | secret | GitHub App private key for the releaser bot |
|
|
32
32
|
|
|
33
|
-
|
|
33
|
+
The trusted publisher on npmjs.com is registered by **file path**, so
|
|
34
34
|
`.github/workflows/release.yml` cannot be renamed or moved without editing that
|
|
35
|
-
registration first.
|
|
36
|
-
nothing earlier in the run reports it.
|
|
35
|
+
registration first. A rename fails the publish with an identity mismatch, and
|
|
36
|
+
nothing earlier in the run reports it. The `release` environment name is bound
|
|
37
37
|
the same way.
|
|
38
38
|
|
|
39
|
-
|
|
39
|
+
Deleting the `release` environment deletes both rows above with it, and there is
|
|
40
40
|
no repo-level fallback: `create-github-app-token` then runs with empty inputs
|
|
41
|
-
and the job fails at that step.
|
|
41
|
+
and the job fails at that step. The private key cannot be read back from
|
|
42
42
|
GitHub; recreating it means generating a new one in the App settings.
|
|
43
43
|
|
|
44
|
-
##
|
|
44
|
+
## Version history
|
|
45
45
|
|
|
46
|
-
|
|
46
|
+
The version and the tag are owned by semantic-release. `tagFormat` is `v${version}` and
|
|
47
47
|
history continues from `v0.1.3`; `v0.1.0`–`v0.1.3` are the legacy git-install
|
|
48
48
|
tags and are never deleted or moved.
|
|
49
49
|
|
|
50
|
-
|
|
50
|
+
During preparation, `@semantic-release/npm` stages the released `package.json`
|
|
51
51
|
version and `@jno21/semantic-release-github-commit` commits it to `main` through
|
|
52
52
|
GitHub's API as the authenticated App. GitHub signs that commit, and the release
|
|
53
|
-
tag points to it.
|
|
53
|
+
tag points to it. The `[skip ci]` marker on that commit is what stops a release
|
|
54
54
|
from releasing itself.
|
|
55
55
|
|
|
56
|
-
|
|
56
|
+
Check what the next version would be without publishing anything:
|
|
57
57
|
|
|
58
58
|
```sh
|
|
59
59
|
pnpm dlx semantic-release --dry-run --no-ci
|
|
60
60
|
```
|
|
61
61
|
|
|
62
|
-
##
|
|
62
|
+
## The artifact
|
|
63
63
|
|
|
64
64
|
`dist/` is generated and untracked. `prepublishOnly` runs `pnpm run verify`,
|
|
65
65
|
which builds it, so the tarball is always packed from a tree that just passed
|
|
66
66
|
the gate. `files` is `dist`, `docs`, `README.md`, `LICENSE`.
|
|
67
67
|
|
|
68
|
-
|
|
68
|
+
Manual publish is emergency recovery only:
|
|
69
69
|
|
|
70
70
|
```sh
|
|
71
71
|
pnpm run verify
|
|
72
72
|
npm publish --access public
|
|
73
73
|
```
|
|
74
74
|
|
|
75
|
-
##
|
|
75
|
+
## The old install path
|
|
76
76
|
|
|
77
|
-
|
|
78
|
-
`github:uinaf/skillcheck#v0.1.3`.
|
|
79
|
-
committed `dist/`, so anything pinned to them keeps working untouched.
|
|
77
|
+
Before `@uinaf/skillcheck` existed, consumers installed
|
|
78
|
+
`github:uinaf/skillcheck#v0.1.3`. Those tags still resolve and still carry a
|
|
79
|
+
committed `dist/`, so anything pinned to them keeps working untouched. New
|
|
80
80
|
consumers use npm.
|
package/docs/scenarios.md
CHANGED
|
@@ -1,20 +1,20 @@
|
|
|
1
|
-
#
|
|
1
|
+
# Writing scenarios
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
A scenario is two files in a frozen location:
|
|
4
4
|
|
|
5
5
|
```text
|
|
6
6
|
<root>/skills/<skill>/evals/<scenario>/task.md
|
|
7
7
|
<root>/skills/<skill>/evals/<scenario>/criteria.json
|
|
8
8
|
```
|
|
9
9
|
|
|
10
|
-
|
|
11
|
-
and the scorecard entry.
|
|
10
|
+
The path is the identity: `<skill>--<scenario>` names the run, the result file,
|
|
11
|
+
and the scorecard entry. On the codex and cursor harnesses the name gains a
|
|
12
12
|
`--codex` or `--cursor` suffix, so every harness can hold results side by side.
|
|
13
|
-
|
|
13
|
+
A directory missing either file is not discovered.
|
|
14
14
|
|
|
15
15
|
## task.md
|
|
16
16
|
|
|
17
|
-
|
|
17
|
+
The prompt handed to the agent, verbatim, with one piece of syntax. Input files
|
|
18
18
|
are embedded inline and materialized into the workdir before the run:
|
|
19
19
|
|
|
20
20
|
```md
|
|
@@ -25,13 +25,13 @@ Fix the failing check in the config below.
|
|
|
25
25
|
======= END FILE =======
|
|
26
26
|
```
|
|
27
27
|
|
|
28
|
-
|
|
29
|
-
is available in your working directory.") and written to disk.
|
|
28
|
+
Each block is replaced in the prompt with a pointer ("Input file `config.json`
|
|
29
|
+
is available in your working directory.") and written to disk. Destinations
|
|
30
30
|
must stay under the workdir, must not collide, and must not target `.claude/`,
|
|
31
31
|
`.agents/`, or `.cursor/`, since a fixture that writes agent config would be
|
|
32
32
|
configuring its own examiner.
|
|
33
33
|
|
|
34
|
-
|
|
34
|
+
Write the task the way a user would write it. Do not name the skill, describe
|
|
35
35
|
its steps, or hint at the checklist: routing is part of what is being measured.
|
|
36
36
|
|
|
37
37
|
## criteria.json
|
|
@@ -46,45 +46,45 @@ its steps, or hint at the checklist: routing is part of what is being measured.
|
|
|
46
46
|
}
|
|
47
47
|
```
|
|
48
48
|
|
|
49
|
-
`type` must be `weighted_checklist` and the checklist must be non-empty.
|
|
49
|
+
`type` must be `weighted_checklist` and the checklist must be non-empty. Every
|
|
50
50
|
item needs a non-empty `name` and `description` and a positive `max_score`.
|
|
51
51
|
|
|
52
|
-
|
|
53
|
-
assert-set with threshold 0.7.
|
|
52
|
+
Each item becomes one `llm-rubric` assertion weighted by `max_score`, inside an
|
|
53
|
+
assert-set with threshold 0.7. A separate `skill-used` assertion sits outside
|
|
54
54
|
that aggregate, so a run that produces good output without ever loading the
|
|
55
|
-
skill still fails.
|
|
55
|
+
skill still fails. There is no test-level threshold: both must pass.
|
|
56
56
|
|
|
57
|
-
|
|
58
|
-
property, not a feeling.
|
|
57
|
+
Write descriptions a judge can check against the deliverable: an observable
|
|
58
|
+
property, not a feeling. Weight the items that would make a reviewer reject the
|
|
59
59
|
work.
|
|
60
60
|
|
|
61
|
-
##
|
|
61
|
+
## What the judge sees
|
|
62
62
|
|
|
63
|
-
|
|
64
|
-
pre-run manifest.
|
|
63
|
+
The agent's final message, plus every file in the workdir that differs from the
|
|
64
|
+
pre-run manifest. Unchanged inputs are omitted; deleted inputs, unreadable
|
|
65
65
|
files, and non-regular files are named rather than read.
|
|
66
66
|
|
|
67
|
-
|
|
68
|
-
appended total at 24,000, with truncation stated inline.
|
|
69
|
-
rubric judges return nothing at all, which is why the caps exist.
|
|
67
|
+
Sections are sorted by path, each file is capped at 4,000 characters and the
|
|
68
|
+
appended total at 24,000, with truncation stated inline. Very large outputs make
|
|
69
|
+
rubric judges return nothing at all, which is why the caps exist. Keep fixtures
|
|
70
70
|
small enough that the deliverable fits.
|
|
71
71
|
|
|
72
|
-
##
|
|
72
|
+
## Hidden skills
|
|
73
73
|
|
|
74
|
-
|
|
75
|
-
production, which the agent SDK cannot simulate.
|
|
74
|
+
A skill with `disable-model-invocation: true` is explicit-invoke-only in
|
|
75
|
+
production, which the agent SDK cannot simulate. So the eval copy, never the shipped one,
|
|
76
76
|
has the flag stripped, and the task gains a leading
|
|
77
|
-
`Use the <skill> skill for this task.`
|
|
78
|
-
behavior-when-invoked rather than routing.
|
|
77
|
+
`Use the <skill> skill for this task.` The eval then measures
|
|
78
|
+
behavior-when-invoked rather than routing. The flag is only honored inside the
|
|
79
79
|
frontmatter block; body text mentioning the key does not count.
|
|
80
80
|
|
|
81
|
-
##
|
|
81
|
+
## The workdir
|
|
82
82
|
|
|
83
|
-
|
|
84
|
-
time.
|
|
83
|
+
Per run, under `<root>/.skillcheck/scratch/<name>/`, rebuilt from scratch each
|
|
84
|
+
time. The skill under test is installed where the harness discovers skills:
|
|
85
85
|
`.claude/skills/<skill>/`, plus `.agents/skills/<skill>/` on codex, or
|
|
86
|
-
`.cursor/skills/<skill>/` alone on cursor
|
|
86
|
+
`.cursor/skills/<skill>/` alone on cursor, with its `evals/` directory
|
|
87
87
|
excluded, so criteria never leak into the agent's context.
|
|
88
88
|
|
|
89
|
-
|
|
89
|
+
Scenario quality is behavioral proof; [authoring](authoring.md) covers the
|
|
90
90
|
judgment layer lint and evals cannot grade.
|
package/docs/usage.md
CHANGED
|
@@ -1,104 +1,120 @@
|
|
|
1
|
-
#
|
|
1
|
+
# Usage
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Every subcommand resolves one root, `--root <dir>` or the current directory.
|
|
4
4
|
`lint` also takes the root as a positional, because that is the shape CI reaches
|
|
5
5
|
for first.
|
|
6
6
|
|
|
7
|
-
##
|
|
7
|
+
## Lint
|
|
8
8
|
|
|
9
9
|
```sh
|
|
10
10
|
skillcheck lint # lints the current repo
|
|
11
11
|
skillcheck lint ../other # lints another root
|
|
12
12
|
```
|
|
13
13
|
|
|
14
|
-
|
|
14
|
+
Checks each `<root>/skills/<skill>/`:
|
|
15
15
|
|
|
16
|
-
-
|
|
17
|
-
-
|
|
16
|
+
- Frontmatter opens with `---` on line 1 and closes
|
|
17
|
+
- Keys are `name`, `description`, `disable-model-invocation` and nothing else,
|
|
18
18
|
each at most once
|
|
19
19
|
- `name` equals the directory name; `description` is non-empty
|
|
20
20
|
- `disable-model-invocation`, when present, is the bare YAML boolean `true`.
|
|
21
|
-
|
|
22
|
-
-
|
|
21
|
+
A quoted `"true"` is an error
|
|
22
|
+
- Relative links in the body resolve on disk
|
|
23
23
|
|
|
24
|
-
|
|
25
|
-
links never fail.
|
|
24
|
+
Code spans and fenced blocks are stripped before links are checked, so example
|
|
25
|
+
links never fail. External schemes and `#anchors` pass. Dot-directories under
|
|
26
26
|
`skills/` (`.claude-plugin`) are plugin metadata, not packages, and are skipped.
|
|
27
27
|
|
|
28
|
-
|
|
28
|
+
Findings print one per line, relative to the linted root, then a count. Exit 0
|
|
29
29
|
clean, 1 with findings.
|
|
30
30
|
|
|
31
|
-
##
|
|
31
|
+
## Run
|
|
32
32
|
|
|
33
33
|
```sh
|
|
34
34
|
skillcheck run skills/<skill>/evals/<scenario>
|
|
35
35
|
skillcheck run <scenario-dir> --agent MODEL --judge MODEL --harness codex --max-turns 80
|
|
36
36
|
```
|
|
37
37
|
|
|
38
|
-
|
|
38
|
+
Materializes the scenario into `<root>/.skillcheck/scratch/<name>/workdir`,
|
|
39
39
|
installs the skill under test into that workdir, drives the agent, and grades
|
|
40
|
-
the files it wrote.
|
|
41
|
-
usable result
|
|
40
|
+
the files it wrote. Exit 0 pass, 1 graded fail, 2 error (promptfoo produced no
|
|
41
|
+
usable result, or the optional eval peers are not installed — the message
|
|
42
|
+
carries the exact `pnpm add` command; see
|
|
43
|
+
[adoption](adoption.md#evals)).
|
|
42
44
|
|
|
43
|
-
|
|
44
|
-
message, and writes no provenance sidecar.
|
|
45
|
+
A test that errored was never graded, so it exits 2, prints the provider's
|
|
46
|
+
message, and writes no provenance sidecar. It is never reported as
|
|
45
47
|
`FAIL score=0.0000`; only a real judged verdict can fail a run.
|
|
46
48
|
|
|
47
|
-
|
|
48
|
-
`--max-turns 50`.
|
|
49
|
+
Defaults: `--harness claude`, agent `claude-opus-5`, judge `claude-opus-5`,
|
|
50
|
+
`--max-turns 50`. On the codex and cursor harnesses, omitting `--agent` leaves
|
|
49
51
|
the model to that CLI's own default.
|
|
50
52
|
|
|
51
53
|
`--harness cursor` drives the scenario through the Cursor Agent CLI
|
|
52
54
|
(`cursor-agent` on PATH) with the skill installed under `.cursor/skills/`;
|
|
53
|
-
`--agent` names a Cursor model id, e.g. `composer-2.5`.
|
|
55
|
+
`--agent` names a Cursor model id, e.g. `composer-2.5`. There is no promptfoo
|
|
54
56
|
cursor provider, so the run uses this package's own provider module, which
|
|
55
57
|
replays the CLI's `stream-json` output: the `result` event becomes the graded
|
|
56
58
|
output and `SKILL.md` reads under `.cursor/skills/` become the `skill-used`
|
|
57
|
-
evidence.
|
|
59
|
+
evidence. The judge leg is unchanged.
|
|
58
60
|
|
|
59
|
-
|
|
61
|
+
`--judge` takes either a bare Claude model (graded through the Anthropic
|
|
62
|
+
selection in [auth](#auth)) or a provider-qualified promptfoo id, passed
|
|
63
|
+
through verbatim:
|
|
64
|
+
|
|
65
|
+
```sh
|
|
66
|
+
skillcheck run <scenario-dir> --judge openai:chat:gpt-5.6-sol --judge-effort high
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
A provider-qualified judge authenticates through that provider's own env
|
|
70
|
+
(`OPENAI_API_KEY`, plus `OPENAI_BASE_URL` for a gateway) and is recorded
|
|
71
|
+
verbatim in the scorecard's `judge_model` column. `--judge-effort`
|
|
72
|
+
(minimal|low|medium|high) sets `reasoning_effort` and requires a
|
|
73
|
+
provider-qualified judge; the Anthropic judge does not take one.
|
|
74
|
+
|
|
75
|
+
## Sweep
|
|
60
76
|
|
|
61
77
|
```sh
|
|
62
78
|
skillcheck sweep # only scenarios without results
|
|
63
79
|
skillcheck sweep --all # rerun everything
|
|
64
80
|
```
|
|
65
81
|
|
|
66
|
-
|
|
67
|
-
order, sequentially.
|
|
68
|
-
discovered.
|
|
82
|
+
Walks `<root>/skills/*/evals/*` and `<root>/cli/*/skills/*/evals/*`, in sorted
|
|
83
|
+
order, sequentially. A scenario needs both `task.md` and `criteria.json` to be
|
|
84
|
+
discovered. Exit 2 if anything errored, 1 if anything failed, else 0.
|
|
69
85
|
|
|
70
|
-
`EVALS_CONCURRENCY` is passed to promptfoo as `-j` (default 4).
|
|
86
|
+
`EVALS_CONCURRENCY` is passed to promptfoo as `-j` (default 4). It parallelizes
|
|
71
87
|
within one scenario, not across them.
|
|
72
88
|
|
|
73
|
-
|
|
89
|
+
One known failure mode: judge calls through a gateway can drop at the transport
|
|
74
90
|
layer ([uinaf/agent-platform#28](https://github.com/uinaf/agent-platform/issues/28)).
|
|
75
|
-
|
|
91
|
+
That surfaces as an ERROR with no usable result, not as a graded FAIL, and the
|
|
76
92
|
mitigation is a rerun. `sweep` without `--all` resumes, so a rerun only picks up
|
|
77
93
|
what is missing.
|
|
78
94
|
|
|
79
|
-
##
|
|
95
|
+
## Summarize
|
|
80
96
|
|
|
81
97
|
```sh
|
|
82
98
|
skillcheck summarize [--allow-mixed]
|
|
83
99
|
```
|
|
84
100
|
|
|
85
|
-
|
|
101
|
+
Reduces `<root>/.skillcheck/results/*.json` into
|
|
86
102
|
`<root>/.skillcheck/scorecards/<UTC-date>.json`: one entry per scenario with
|
|
87
103
|
skill, scenario, harness, tree sha, score, pass, both models, latency, tokens.
|
|
88
104
|
|
|
89
|
-
|
|
105
|
+
If a scorecard for today already exists, the two are merged on
|
|
90
106
|
`(skill, scenario, harness)`: entries from this run win, entries it did not
|
|
91
|
-
touch survive, and the merge is reported on stdout.
|
|
107
|
+
touch survive, and the merge is reported on stdout. Summarizing after rerunning
|
|
92
108
|
six of twenty-nine scenarios therefore leaves twenty-nine rows in the file, not
|
|
93
|
-
six.
|
|
109
|
+
six. A same-date file that cannot be parsed stops the write instead of being
|
|
94
110
|
overwritten.
|
|
95
111
|
|
|
96
|
-
|
|
112
|
+
Files that are not promptfoo results are skipped with a warning rather than
|
|
97
113
|
failing the reduction.
|
|
98
114
|
|
|
99
|
-
##
|
|
115
|
+
## Provenance
|
|
100
116
|
|
|
101
|
-
|
|
117
|
+
Each successful run writes a `<name>.meta.json` sidecar next to its result:
|
|
102
118
|
|
|
103
119
|
```json
|
|
104
120
|
{
|
|
@@ -111,24 +127,26 @@ each successful run writes a `<name>.meta.json` sidecar next to its result:
|
|
|
111
127
|
|
|
112
128
|
`summarize` reads those sidecars and refuses to mix skills-tree revisions in one
|
|
113
129
|
scorecard unless `--allow-mixed`, in which case the top-level `skills_tree_sha`
|
|
114
|
-
becomes `mixed` and per-entry shas remain.
|
|
130
|
+
becomes `mixed` and per-entry shas remain. A result with no sidecar reduces as
|
|
115
131
|
`unattested`.
|
|
116
132
|
|
|
117
|
-
##
|
|
133
|
+
## State
|
|
118
134
|
|
|
119
135
|
`<root>/.skillcheck/` holds `scratch/` and `results/`, both disposable and safe
|
|
120
|
-
to gitignore, and `scorecards/`, which is meant to be committed.
|
|
136
|
+
to gitignore, and `scorecards/`, which is meant to be committed. Nothing is ever
|
|
121
137
|
written inside the installed package.
|
|
122
138
|
|
|
123
|
-
##
|
|
139
|
+
## Auth
|
|
124
140
|
|
|
125
|
-
|
|
|
141
|
+
| Variable | Effect |
|
|
126
142
|
| --------------------------------------------- | ----------------------------------------------------------------- |
|
|
127
|
-
| `ANTHROPIC_BASE_URL` + `ANTHROPIC_AUTH_TOKEN` | claude agent and judge go through a gateway
|
|
128
|
-
|
|
|
129
|
-
| `ANTHROPIC_API_KEY` |
|
|
130
|
-
| `CODEX_HOME` (default `~/.codex`) |
|
|
131
|
-
| `OPENAI_API_KEY` |
|
|
132
|
-
| `CURSOR_API_KEY` |
|
|
133
|
-
|
|
134
|
-
|
|
143
|
+
| `ANTHROPIC_BASE_URL` + `ANTHROPIC_AUTH_TOKEN` | The claude agent and judge go through a gateway |
|
|
144
|
+
| None of the above | Falls back to the local Claude Code session |
|
|
145
|
+
| `ANTHROPIC_API_KEY` | Judge grades over `anthropic:messages:<model>` instead of the SDK |
|
|
146
|
+
| `CODEX_HOME` (default `~/.codex`) | Where the codex harness finds the local `codex` CLI login |
|
|
147
|
+
| `OPENAI_API_KEY` | Agent auth for codex when there is no local login |
|
|
148
|
+
| `CURSOR_API_KEY` | Agent auth for cursor; a logged-in `cursor-agent` also works |
|
|
149
|
+
| `OPENAI_API_KEY` + `OPENAI_BASE_URL` | A provider-qualified `--judge openai:…`, optionally via a gateway |
|
|
150
|
+
|
|
151
|
+
A bare `--judge` model stays on the Anthropic selection regardless of the
|
|
152
|
+
agent harness; a provider-qualified `--judge` uses that provider's env instead.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@uinaf/skillcheck",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.5.0",
|
|
4
4
|
"description": "Lint and eval harness for agent skills",
|
|
5
5
|
"homepage": "https://github.com/uinaf/skillcheck#readme",
|
|
6
6
|
"bugs": {
|
|
@@ -31,16 +31,30 @@
|
|
|
31
31
|
"prepare": "vp config --no-agent",
|
|
32
32
|
"prepublishOnly": "pnpm run verify"
|
|
33
33
|
},
|
|
34
|
-
"
|
|
34
|
+
"devDependencies": {
|
|
35
35
|
"@anthropic-ai/claude-agent-sdk": "^0.3.233",
|
|
36
36
|
"@openai/codex-sdk": "^0.147.0",
|
|
37
|
-
"promptfoo": "^0.122.0"
|
|
38
|
-
},
|
|
39
|
-
"devDependencies": {
|
|
40
37
|
"@types/node": "^26.2.0",
|
|
38
|
+
"promptfoo": "^0.122.0",
|
|
41
39
|
"vite": "catalog:",
|
|
42
40
|
"vite-plus": "catalog:"
|
|
43
41
|
},
|
|
42
|
+
"peerDependencies": {
|
|
43
|
+
"@anthropic-ai/claude-agent-sdk": "^0.3.233",
|
|
44
|
+
"@openai/codex-sdk": "^0.147.0",
|
|
45
|
+
"promptfoo": "^0.122.0"
|
|
46
|
+
},
|
|
47
|
+
"peerDependenciesMeta": {
|
|
48
|
+
"@anthropic-ai/claude-agent-sdk": {
|
|
49
|
+
"optional": true
|
|
50
|
+
},
|
|
51
|
+
"@openai/codex-sdk": {
|
|
52
|
+
"optional": true
|
|
53
|
+
},
|
|
54
|
+
"promptfoo": {
|
|
55
|
+
"optional": true
|
|
56
|
+
}
|
|
57
|
+
},
|
|
44
58
|
"engines": {
|
|
45
59
|
"node": ">=24"
|
|
46
60
|
},
|