@icntswm/skillcheck 1.1.0 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +73 -88
- package/dist/agents/claude.js +8 -5
- package/dist/baseline.js +59 -0
- package/dist/cli.js +309 -26
- package/dist/gen.js +119 -0
- package/dist/import.js +53 -0
- package/dist/junit.js +6 -0
- package/dist/markdown.js +21 -2
- package/dist/report.js +15 -2
- package/dist/results.js +8 -3
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -8,50 +8,49 @@
|
|
|
8
8
|
**Regression tests for Claude Code skills.**
|
|
9
9
|
|
|
10
10
|
You edited a skill description, installed a plugin or switched models, and now
|
|
11
|
-
some
|
|
11
|
+
some requests load the wrong skill. Nothing warns you: the agent still
|
|
12
12
|
answers, just with the wrong instructions. skillcheck catches this before your
|
|
13
13
|
users do.
|
|
14
14
|
|
|
15
15
|

|
|
16
16
|
|
|
17
|
-
A real run on the [demo](examples/demo): one
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
requests.
|
|
17
|
+
<sub>A real run on the [demo](examples/demo): one description got wider and
|
|
18
|
+
started taking its neighbour's requests. The free `lint` flags it, one batch
|
|
19
|
+
call finds both misrouted requests.</sub>
|
|
21
20
|
|
|
22
|
-
##
|
|
21
|
+
## Quick start
|
|
22
|
+
|
|
23
|
+
Needs Node 20+ and [Claude Code](https://docs.anthropic.com/en/docs/claude-code)
|
|
24
|
+
(`claude`) in `PATH`, logged in.
|
|
25
|
+
|
|
26
|
+
```sh
|
|
27
|
+
npm install -g @icntswm/skillcheck
|
|
28
|
+
|
|
29
|
+
skillcheck init # skillcheck.yaml listing your skills; add real requests
|
|
30
|
+
skillcheck lint # free static checks, no model calls
|
|
31
|
+
skillcheck run --batch # cheap pre-check: one call per 25 cases
|
|
32
|
+
skillcheck run --only 2,5 # confirm what the batch flagged with real runs
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
| Command | What it does |
|
|
36
|
+
|---|---|
|
|
37
|
+
| `init` | starter `skillcheck.yaml` with your skill names, free |
|
|
38
|
+
| `gen` | the model drafts cases from your descriptions, for you to review |
|
|
39
|
+
| `import` | turns a skill-creator trigger eval set into cases |
|
|
40
|
+
| `check` | validates the file and catches misspelled or renamed skills, free |
|
|
41
|
+
| `lint` | short or look-alike descriptions, free |
|
|
42
|
+
| `list` | the skills and commands Claude Code sees, free |
|
|
43
|
+
| `run` | runs the cases; `--batch` for the cheap pre-check |
|
|
44
|
+
|
|
45
|
+
Every option, the default file lookup and exit codes are in
|
|
46
|
+
[Commands](docs/commands.md).
|
|
23
47
|
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
- 🔍 **Points at the fix.** The confusion block shows which skill took whose
|
|
31
|
-
requests. When the model names the right skill but doesn't load it, the
|
|
32
|
-
report says so separately.
|
|
33
|
-
- 🛡️ **Safe on any project.** Runs go in plan mode with edits, shell, web and
|
|
34
|
-
MCP tools disabled. Your files are never touched.
|
|
35
|
-
- 🧪 **Honest about randomness.** `--repeat` and `threshold` tell a flaky case
|
|
36
|
-
from a broken one instead of letting you guess.
|
|
37
|
-
- 🏷️ **Catches renames and typos.** Case names are checked against the skills
|
|
38
|
-
the agent really has, so a renamed skill can't pass silently.
|
|
39
|
-
- ⚙️ **CI-ready.** A GitHub Action (`uses: icntswm/skillcheck@v1`) that
|
|
40
|
-
comments the report on the pull request, JUnit, JSON and Markdown reports,
|
|
41
|
-
clear exit codes, a spending cap (`--budget`), and `--config-dir` to test
|
|
42
|
-
only the skills in your repository.
|
|
43
|
-
- 📄 **Plain YAML, one dependency.** Cases are readable by anyone on the team
|
|
44
|
-
and live next to the skills they test.
|
|
45
|
-
|
|
46
|
-
## When to run it
|
|
47
|
-
|
|
48
|
-
- after you rewrite or shorten a skill description;
|
|
49
|
-
- when you add a skill that sounds like an existing one;
|
|
50
|
-
- after installing a plugin that brings its own skills;
|
|
51
|
-
- before switching to another model;
|
|
52
|
-
- on every pull request that touches `skills/`.
|
|
53
|
-
|
|
54
|
-
## How it works in one minute
|
|
48
|
+
No cases yet? `skillcheck gen -o skillcheck.yaml` lets the model draft them from
|
|
49
|
+
your skill descriptions, for you to review. Coming from Anthropic's
|
|
50
|
+
skill-creator? `skillcheck import eval_set.json --skill <name>` converts its
|
|
51
|
+
trigger eval set. To try without installing: `npx @icntswm/skillcheck lint`.
|
|
52
|
+
|
|
53
|
+
## How it works
|
|
55
54
|
|
|
56
55
|
Write requests the way you actually type them, and say which skill must, or
|
|
57
56
|
must not, load:
|
|
@@ -66,80 +65,66 @@ cases:
|
|
|
66
65
|
```
|
|
67
66
|
|
|
68
67
|
skillcheck sends each request to a headless `claude -p`, watches which skills
|
|
69
|
-
the model loads, and stops the process as soon as the answer is clear.
|
|
70
|
-
|
|
71
|
-
|
|
68
|
+
the model loads, and stops the process as soon as the answer is clear. Runs go
|
|
69
|
+
in plan mode with edits, shell, web and MCP tools off, so your files are never
|
|
70
|
+
touched.
|
|
72
71
|
|
|
73
|
-
|
|
72
|
+
Three levels, from free to exact. Go up only when the cheaper one finds nothing:
|
|
74
73
|
|
|
75
74
|
| Command | Model calls | What it tells you |
|
|
76
75
|
|---|---|---|
|
|
77
|
-
| `skillcheck lint` | none | short
|
|
76
|
+
| `skillcheck lint` | none | short or look-alike descriptions, cases that share few words with their skill |
|
|
78
77
|
| `skillcheck run --batch` | one per 25 cases | which skill the model *says* it would load |
|
|
79
78
|
| `skillcheck run` | one per case | which skill the model *actually* loads |
|
|
80
79
|
|
|
81
|
-
On the demo
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
Yes, and the [demo](examples/demo) shows it: six skills, twelve cases and a bad
|
|
87
|
-
edit of two descriptions.
|
|
80
|
+
On the [demo](examples/demo) (six skills, twelve cases, a bad edit of two
|
|
81
|
+
descriptions) `lint` flagged the short description, and both `run --batch` and
|
|
82
|
+
`run` caught the two misrouted cases. One batch call cost $0.05–0.10 and agreed
|
|
83
|
+
with the full run in 34 checks out of 34.
|
|
88
84
|
|
|
89
|
-
|
|
90
|
-
|---|---|---|
|
|
91
|
-
| `lint` | no problems | flags the too-short description |
|
|
92
|
-
| `run --batch` | 12/12 passed | catches both misrouted cases |
|
|
93
|
-
| `run` | 5/5 passed | catches the same two cases |
|
|
94
|
-
|
|
95
|
-
The demo README also lists edits that did *not* break routing, and why.
|
|
96
|
-
|
|
97
|
-
## Install
|
|
98
|
-
|
|
99
|
-
Needs Node 20+ and [Claude Code](https://docs.anthropic.com/en/docs/claude-code)
|
|
100
|
-
(`claude`) in `PATH`, logged in.
|
|
101
|
-
|
|
102
|
-
```
|
|
103
|
-
npm install -g @icntswm/skillcheck
|
|
104
|
-
```
|
|
85
|
+
## What you get
|
|
105
86
|
|
|
106
|
-
The
|
|
107
|
-
|
|
87
|
+
- 🎯 **The real routing decision**, read from Claude Code itself, not guessed from keywords.
|
|
88
|
+
- 💸 **You pay for the decision, not the work**: a run stops the moment a skill is picked.
|
|
89
|
+
- 🔍 **Points at the fix**: the confusion block shows which skill took whose requests.
|
|
90
|
+
- 🧪 **Honest about randomness**: `--repeat` and `threshold` tell a flaky case from a broken one.
|
|
91
|
+
- 🏷️ **Catches renames**: case names are checked against the skills the agent really has.
|
|
92
|
+
- 📄 **Plain YAML, one dependency**: cases live next to the skills they test.
|
|
108
93
|
|
|
109
|
-
##
|
|
94
|
+
## In CI
|
|
110
95
|
|
|
111
|
-
```
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
96
|
+
```yaml
|
|
97
|
+
- uses: icntswm/skillcheck@v1
|
|
98
|
+
env:
|
|
99
|
+
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
|
100
|
+
with:
|
|
101
|
+
model: sonnet
|
|
102
|
+
budget: 2
|
|
118
103
|
```
|
|
119
104
|
|
|
120
|
-
|
|
121
|
-
|
|
105
|
+
The action tests only the skills in your repository, comments the report on the
|
|
106
|
+
pull request and writes JUnit, JSON and Markdown reports. `--budget` caps the
|
|
107
|
+
spend; `--baseline` with `--only-new-failures` compares with `main` and fails
|
|
108
|
+
only on new failures. Run it after you rewrite a description, add a skill that
|
|
109
|
+
sounds like an existing one, install a plugin, or switch models.
|
|
122
110
|
|
|
123
111
|
## Documentation
|
|
124
112
|
|
|
125
113
|
| | |
|
|
126
114
|
|---|---|
|
|
127
|
-
| [
|
|
115
|
+
| [Commands](docs/commands.md) | every command and option, exit codes |
|
|
116
|
+
| [Writing cases](docs/writing-cases.md) | the file format, what makes a good case, `gen` and `import` |
|
|
128
117
|
| [Cost](docs/cost.md) | what a run costs and how to spend less |
|
|
129
|
-
| [Reports and CI](docs/ci.md) | confusion block,
|
|
118
|
+
| [Reports and CI](docs/ci.md) | confusion block, reports, GitHub Actions, comparing with `main` |
|
|
130
119
|
| [How it works](docs/how-it-works.md) | what happens inside a run, and the limits of each level |
|
|
131
120
|
|
|
132
|
-
`skillcheck --help` lists every command and option.
|
|
133
|
-
|
|
134
121
|
## Status
|
|
135
122
|
|
|
136
|
-
Works with Claude Code
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
languages that put spaces between words, and is of little use for Chinese,
|
|
142
|
-
Japanese or Korean.
|
|
123
|
+
Works with Claude Code; other agents with skills can be added behind a small
|
|
124
|
+
adapter interface. Requests can be in any language: `run` asks the model
|
|
125
|
+
itself. `lint` compares words, so it is tuned for English and Russian, works
|
|
126
|
+
roughly for other languages with spaces between words, and is of little use for
|
|
127
|
+
Chinese, Japanese or Korean.
|
|
143
128
|
|
|
144
129
|
## License
|
|
145
130
|
|
package/dist/agents/claude.js
CHANGED
|
@@ -327,7 +327,8 @@ export class ClaudeAdapter {
|
|
|
327
327
|
},
|
|
328
328
|
});
|
|
329
329
|
const { text, costUsd, structuredOutput } = out.stream.result;
|
|
330
|
-
|
|
330
|
+
const error = withLoginHint(out.error, opts.configDir);
|
|
331
|
+
return { structured: structuredOutput, text, costUsd, usage: out.stream.usage, error, durationMs: out.durationMs };
|
|
331
332
|
}
|
|
332
333
|
finally {
|
|
333
334
|
await rm(workdir, { recursive: true, force: true });
|
|
@@ -408,13 +409,15 @@ export class ClaudeAdapter {
|
|
|
408
409
|
clearTimeout(killTimer);
|
|
409
410
|
if (timeoutTimer)
|
|
410
411
|
clearTimeout(timeoutTimer);
|
|
411
|
-
// A non-zero exit is not a failure by itself: error_max_turns is normal
|
|
412
|
-
//
|
|
412
|
+
// A non-zero exit is not a failure by itself: error_max_turns is normal
|
|
413
|
+
// and ends with a result event. Without one, the run ended before routing
|
|
414
|
+
// was over, unless we stopped it ourselves.
|
|
413
415
|
if (!error)
|
|
414
416
|
error = stream.error;
|
|
415
|
-
if (!error && !
|
|
417
|
+
if (!error && !stopping && !stream.finished) {
|
|
416
418
|
const detail = stderrText.trim().slice(0, 120);
|
|
417
|
-
|
|
419
|
+
const why = stdoutSeen ? `claude exited with code ${code ?? signal} before its result` : `claude exited with code ${code ?? signal}`;
|
|
420
|
+
error = `${why}${detail ? `: ${detail}` : ""}`;
|
|
418
421
|
}
|
|
419
422
|
resolve({ stream, error, stoppedEarly, durationMs: Date.now() - startedAt });
|
|
420
423
|
};
|
package/dist/baseline.js
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
const STATUSES = ["passed", "failed", "skipped"];
|
|
2
|
+
/** Parse a --json report; checked up front so a bad file fails before any model call. */
|
|
3
|
+
export function parseBaseline(text) {
|
|
4
|
+
const value = JSON.parse(text);
|
|
5
|
+
if (value === null || typeof value !== "object" || value.tool !== "skillcheck" || !Array.isArray(value.cases)) {
|
|
6
|
+
throw new Error("not a skillcheck --json report");
|
|
7
|
+
}
|
|
8
|
+
value.cases.forEach((c, i) => {
|
|
9
|
+
const entry = c;
|
|
10
|
+
if (typeof entry !== "object" || entry === null || typeof entry.query !== "string" || !STATUSES.includes(entry.status) ||
|
|
11
|
+
(entry.id !== null && typeof entry.id !== "string")) {
|
|
12
|
+
throw new Error(`case ${i + 1} is not a report case (needs query, id and status passed, failed or skipped)`);
|
|
13
|
+
}
|
|
14
|
+
});
|
|
15
|
+
return value;
|
|
16
|
+
}
|
|
17
|
+
/**
|
|
18
|
+
* Compare current cases with a previous report. suite is every case of the cases file, so cases
|
|
19
|
+
* left out by --only or --skill do not count as removed.
|
|
20
|
+
*/
|
|
21
|
+
export function compare(current, baseline, file, suite = current) {
|
|
22
|
+
const previous = new Map();
|
|
23
|
+
for (const c of baseline.cases) {
|
|
24
|
+
const key = caseKey(c);
|
|
25
|
+
if (!previous.has(key))
|
|
26
|
+
previous.set(key, c);
|
|
27
|
+
}
|
|
28
|
+
let regressed = 0;
|
|
29
|
+
let fixed = 0;
|
|
30
|
+
let newCount = 0;
|
|
31
|
+
const changes = current.map((c) => {
|
|
32
|
+
const old = previous.get(caseKey(c));
|
|
33
|
+
if (c.status === "skipped")
|
|
34
|
+
return null;
|
|
35
|
+
// a case the baseline budget skipped has no result to compare with
|
|
36
|
+
if (old === undefined || old.status === "skipped") {
|
|
37
|
+
newCount++;
|
|
38
|
+
return "new";
|
|
39
|
+
}
|
|
40
|
+
if (old.status === "passed" && c.status === "failed") {
|
|
41
|
+
regressed++;
|
|
42
|
+
return "regressed";
|
|
43
|
+
}
|
|
44
|
+
if (old.status === "failed" && c.status === "passed") {
|
|
45
|
+
fixed++;
|
|
46
|
+
return "fixed";
|
|
47
|
+
}
|
|
48
|
+
return null;
|
|
49
|
+
});
|
|
50
|
+
const suiteKeys = new Set(suite.map(caseKey));
|
|
51
|
+
let removed = 0;
|
|
52
|
+
for (const key of previous.keys())
|
|
53
|
+
if (!suiteKeys.has(key))
|
|
54
|
+
removed++;
|
|
55
|
+
return { changes, summary: { file, regressed, fixed, new: newCount, removed } };
|
|
56
|
+
}
|
|
57
|
+
function caseKey(c) {
|
|
58
|
+
return c.id !== null ? `id:${c.id}` : `query:${c.query}`;
|
|
59
|
+
}
|
package/dist/cli.js
CHANGED
|
@@ -4,13 +4,17 @@ import { readFile } from "node:fs/promises";
|
|
|
4
4
|
import * as path from "node:path";
|
|
5
5
|
import { pathToFileURL } from "node:url";
|
|
6
6
|
import { parseArgs } from "node:util";
|
|
7
|
+
import { isSeq, parse as parseYaml, parseDocument } from "yaml";
|
|
7
8
|
import { abortActiveRuns, ClaudeAdapter, DEFAULT_DIRECTIVE } from "./agents/claude.js";
|
|
8
9
|
import { BATCH_SCHEMA, buildBatchPrompt, parseBatchAnswer } from "./batch.js";
|
|
10
|
+
import { parseBaseline } from "./baseline.js";
|
|
9
11
|
import { Budget } from "./budget.js";
|
|
10
|
-
import { BUILTIN_SKILLS, ConfigError, findDefaultCasesFile, knownSkillNames, loadSuite, unknownNames, } from "./cases.js";
|
|
12
|
+
import { BUILTIN_SKILLS, ConfigError, findDefaultCasesFile, knownSkillNames, loadSuite, unknownNames, parseSuite, } from "./cases.js";
|
|
11
13
|
import { confusion } from "./confusion.js";
|
|
12
14
|
import { loadSkillDocs } from "./describe.js";
|
|
15
|
+
import { GEN_SCHEMA, buildGenPrompt, caseEntry, genSuite, orderCases, parseGenAnswer } from "./gen.js";
|
|
13
16
|
import { aggregate, judge } from "./judge.js";
|
|
17
|
+
import { evalSetToSuite } from "./import.js";
|
|
14
18
|
import { toJunit } from "./junit.js";
|
|
15
19
|
import { lint, LINT_DEFAULTS } from "./lint.js";
|
|
16
20
|
import { toMarkdown } from "./markdown.js";
|
|
@@ -26,6 +30,9 @@ Usage:
|
|
|
26
30
|
skillcheck lint [file] [options] static checks on descriptions and cases
|
|
27
31
|
skillcheck list [options] print the skills and commands the agent can load
|
|
28
32
|
skillcheck init [file] [options] write a starter cases file with those names
|
|
33
|
+
skillcheck import <file> --skill <name>
|
|
34
|
+
turn a skill-creator trigger eval set into cases
|
|
35
|
+
skillcheck gen [options] draft cases from skill descriptions (one model call per 8 skills)
|
|
29
36
|
|
|
30
37
|
run options:
|
|
31
38
|
-a, --agent <name> agent to route with (default: suite or claude)
|
|
@@ -44,6 +51,8 @@ run options:
|
|
|
44
51
|
and the terminal report to stderr)
|
|
45
52
|
--junit <path> JUnit XML report (GitLab/GitHub test reporters) to path
|
|
46
53
|
--markdown <path> Markdown summary (PR comments, GitHub job summary) to path
|
|
54
|
+
--baseline <path> earlier --json report: mark regressed, fixed and new cases
|
|
55
|
+
--only-new-failures exit 1 only for regressed or new failing cases, or budget skips (needs --baseline)
|
|
47
56
|
run/check options:
|
|
48
57
|
--skill <a,b> keep only cases that mention these skills
|
|
49
58
|
check options:
|
|
@@ -59,6 +68,18 @@ list/init options:
|
|
|
59
68
|
CLAUDE_CONFIG_DIR; run, list and init also set it for the agent
|
|
60
69
|
init options:
|
|
61
70
|
--force overwrite an existing file
|
|
71
|
+
import options:
|
|
72
|
+
--skill <name> the skill the eval set is about
|
|
73
|
+
-o, --out <file> write the cases there instead of stdout (--force overwrites)
|
|
74
|
+
|
|
75
|
+
gen options:
|
|
76
|
+
--skill <a,b> skills to draft (default: user/project skills)
|
|
77
|
+
--plugin <name> skills of this installed plugin
|
|
78
|
+
--per-skill <n> positive requests per skill (default: 4)
|
|
79
|
+
-o, --out <file> write the cases there instead of stdout (--force overwrites)
|
|
80
|
+
--append <file> add the draft to an existing cases file (default: skills it lacks)
|
|
81
|
+
-m, -j, --timeout and --config-dir work as for run
|
|
82
|
+
|
|
62
83
|
common:
|
|
63
84
|
-h, --help show this help
|
|
64
85
|
--version show version
|
|
@@ -69,11 +90,12 @@ nothing reaches the model and nothing is billed, even when not logged in.
|
|
|
69
90
|
|
|
70
91
|
Exit codes: 0 all passed, 1 some case failed or was skipped, 2 config or environment error.
|
|
71
92
|
`;
|
|
93
|
+
const COMMANDS = ["run", "check", "lint", "list", "init", "import", "gen"];
|
|
72
94
|
const OPTIONS = {
|
|
73
95
|
help: { type: "boolean", short: "h", default: false },
|
|
74
96
|
version: { type: "boolean", default: false },
|
|
75
|
-
agent: { type: "string" },
|
|
76
|
-
model: { type: "string" },
|
|
97
|
+
agent: { type: "string", short: "a" },
|
|
98
|
+
model: { type: "string", short: "m" },
|
|
77
99
|
jobs: { type: "string", short: "j" },
|
|
78
100
|
only: { type: "string" },
|
|
79
101
|
repeat: { type: "string" },
|
|
@@ -89,21 +111,34 @@ const OPTIONS = {
|
|
|
89
111
|
json: { type: "string" },
|
|
90
112
|
junit: { type: "string" },
|
|
91
113
|
markdown: { type: "string" },
|
|
114
|
+
baseline: { type: "string" },
|
|
115
|
+
"only-new-failures": { type: "boolean", default: false },
|
|
92
116
|
top: { type: "string" },
|
|
93
117
|
overlap: { type: "string" },
|
|
94
118
|
strict: { type: "boolean", default: false },
|
|
95
119
|
"config-dir": { type: "string" },
|
|
96
120
|
force: { type: "boolean", default: false },
|
|
121
|
+
out: { type: "string", short: "o" },
|
|
122
|
+
"per-skill": { type: "string" },
|
|
123
|
+
plugin: { type: "string" },
|
|
124
|
+
append: { type: "string" },
|
|
97
125
|
};
|
|
98
126
|
/** Config/environment problems: message to stderr, exit 2. */
|
|
99
127
|
class UsageError extends Error {
|
|
100
128
|
}
|
|
101
129
|
function parseFlags(values, cwd) {
|
|
130
|
+
if (values.plugin !== undefined && values.skill !== undefined)
|
|
131
|
+
throw new UsageError("--plugin and --skill do not go together");
|
|
132
|
+
if (values.append !== undefined && values.out !== undefined)
|
|
133
|
+
throw new UsageError("--append and --out do not go together");
|
|
102
134
|
if (values["batch-size"] !== undefined && !values.batch)
|
|
103
135
|
throw new UsageError("--batch-size requires --batch");
|
|
104
136
|
if (values.batch && (values.directive !== undefined || values["no-early-stop"])) {
|
|
105
137
|
throw new UsageError("--directive and --no-early-stop do not apply to --batch");
|
|
106
138
|
}
|
|
139
|
+
if (values["only-new-failures"] && values.baseline === undefined) {
|
|
140
|
+
throw new UsageError("--only-new-failures requires --baseline");
|
|
141
|
+
}
|
|
107
142
|
return {
|
|
108
143
|
help: values.help,
|
|
109
144
|
version: values.version,
|
|
@@ -120,15 +155,21 @@ function parseFlags(values, cwd) {
|
|
|
120
155
|
batch: values.batch,
|
|
121
156
|
batchSize: intFlag(values["batch-size"], "batch-size", 1) ?? 25,
|
|
122
157
|
skill: values.skill,
|
|
158
|
+
plugin: values.plugin,
|
|
159
|
+
append: values.append,
|
|
123
160
|
budget: numberFlag(values.budget),
|
|
124
161
|
json: values.json,
|
|
125
162
|
junit: values.junit,
|
|
126
163
|
markdown: values.markdown,
|
|
164
|
+
baseline: values.baseline,
|
|
165
|
+
onlyNewFailures: values["only-new-failures"],
|
|
127
166
|
top: intFlag(values.top, "top", 1) ?? 5,
|
|
128
167
|
overlap: ratioFlag(values.overlap) ?? 0.3,
|
|
129
168
|
strict: values.strict,
|
|
130
169
|
configDir: values["config-dir"] !== undefined ? resolveConfigDir(values["config-dir"], cwd) : undefined,
|
|
131
170
|
force: values.force,
|
|
171
|
+
out: values.out,
|
|
172
|
+
perSkill: intFlag(values["per-skill"], "per-skill", 1) ?? 4,
|
|
132
173
|
};
|
|
133
174
|
}
|
|
134
175
|
/** --config-dir: absolute, must be an existing directory. */
|
|
@@ -203,7 +244,7 @@ export async function main(argv, io = { stdout: process.stdout, stderr: process.
|
|
|
203
244
|
io.stderr.write(USAGE);
|
|
204
245
|
return 2;
|
|
205
246
|
}
|
|
206
|
-
if (command
|
|
247
|
+
if (!COMMANDS.includes(command)) {
|
|
207
248
|
io.stderr.write(`skillcheck: unknown command "${command}"\n\n${USAGE}`);
|
|
208
249
|
return 2;
|
|
209
250
|
}
|
|
@@ -214,6 +255,10 @@ export async function main(argv, io = { stdout: process.stdout, stderr: process.
|
|
|
214
255
|
return await listCommand(flags, io, deps);
|
|
215
256
|
if (command === "init")
|
|
216
257
|
return await initCommand(positionals[1], flags, io, deps);
|
|
258
|
+
if (command === "import")
|
|
259
|
+
return importCommand(positionals[1], flags, io);
|
|
260
|
+
if (command === "gen")
|
|
261
|
+
return await genCommand(flags, io, deps);
|
|
217
262
|
const file = casesFile(positionals[1], io);
|
|
218
263
|
if (command === "run")
|
|
219
264
|
return await runCommand(file, flags, io, deps);
|
|
@@ -376,6 +421,205 @@ async function initCommand(positional, flags, io, deps) {
|
|
|
376
421
|
io.stdout.write(`wrote ${file} (${skills.length} skills listed)\n`);
|
|
377
422
|
return 0;
|
|
378
423
|
}
|
|
424
|
+
function importCommand(positional, flags, io) {
|
|
425
|
+
if (positional === undefined) {
|
|
426
|
+
throw new UsageError("import needs the eval set file: skillcheck import <eval_set.json> --skill <name>");
|
|
427
|
+
}
|
|
428
|
+
const skill = flags.skill?.trim();
|
|
429
|
+
if (!skill || skill.includes(","))
|
|
430
|
+
throw new UsageError("import needs --skill <name>: the skill the eval set is about");
|
|
431
|
+
const source = path.resolve(io.cwd, positional);
|
|
432
|
+
let data;
|
|
433
|
+
try {
|
|
434
|
+
data = JSON.parse(fs.readFileSync(source, "utf8"));
|
|
435
|
+
}
|
|
436
|
+
catch (e) {
|
|
437
|
+
throw new UsageError(`cannot read ${positional}: ${e.message}`);
|
|
438
|
+
}
|
|
439
|
+
let suite;
|
|
440
|
+
try {
|
|
441
|
+
suite = evalSetToSuite(data, skill, source);
|
|
442
|
+
}
|
|
443
|
+
catch (e) {
|
|
444
|
+
if (e instanceof ConfigError)
|
|
445
|
+
throw new UsageError(`${positional}: ${e.errors.join("; ")}`);
|
|
446
|
+
throw e;
|
|
447
|
+
}
|
|
448
|
+
if (flags.out === undefined) {
|
|
449
|
+
io.stdout.write(suite.text);
|
|
450
|
+
return 0;
|
|
451
|
+
}
|
|
452
|
+
const out = path.resolve(io.cwd, flags.out);
|
|
453
|
+
if (fs.existsSync(out) && !flags.force)
|
|
454
|
+
throw new UsageError(`${flags.out} exists, use --force to overwrite`);
|
|
455
|
+
try {
|
|
456
|
+
fs.writeFileSync(out, suite.text);
|
|
457
|
+
}
|
|
458
|
+
catch (e) {
|
|
459
|
+
throw new UsageError(`cannot write ${flags.out}: ${e.message}`);
|
|
460
|
+
}
|
|
461
|
+
const total = suite.positive + suite.negative;
|
|
462
|
+
io.stdout.write(`wrote ${flags.out} (${total} cases: ${suite.positive} should load ${skill}, ${suite.negative} should not)\n`);
|
|
463
|
+
return 0;
|
|
464
|
+
}
|
|
465
|
+
async function genCommand(flags, io, deps) {
|
|
466
|
+
const docs = loadSkillDocs({ cwd: io.cwd, configDir: flags.configDir });
|
|
467
|
+
const byName = new Map(docs.map((doc) => [doc.name, doc]));
|
|
468
|
+
const append = flags.append === undefined ? null : path.resolve(io.cwd, flags.append);
|
|
469
|
+
const existing = append === null ? null : await loadCases(append, io);
|
|
470
|
+
if (append !== null && existing === null)
|
|
471
|
+
return 2;
|
|
472
|
+
let targets;
|
|
473
|
+
if (flags.skill !== undefined) {
|
|
474
|
+
const names = flags.skill.split(",").map((name) => name.trim()).filter((name) => name !== "");
|
|
475
|
+
const unknown = names.filter((name, i) => !byName.has(name) && names.indexOf(name) === i);
|
|
476
|
+
if (unknown.length > 0)
|
|
477
|
+
throw new UsageError(`unknown skill: ${unknown.join(", ")}`);
|
|
478
|
+
targets = names.map((name) => byName.get(name)).filter((doc, i, all) => all.findIndex((other) => other.name === doc.name) === i);
|
|
479
|
+
}
|
|
480
|
+
else if (flags.plugin !== undefined) {
|
|
481
|
+
const pluginDocs = docs.filter((doc) => doc.plugin === flags.plugin);
|
|
482
|
+
if (pluginDocs.length === 0) {
|
|
483
|
+
const installed = [...new Set(docs.flatMap((doc) => doc.plugin === null ? [] : [doc.plugin]))].sort();
|
|
484
|
+
throw new UsageError(`unknown plugin: ${flags.plugin} (installed: ${installed.length > 0 ? installed.join(", ") : "none installed"})`);
|
|
485
|
+
}
|
|
486
|
+
targets = pluginDocs.filter((doc) => doc.kind === "skill");
|
|
487
|
+
if (targets.length === 0)
|
|
488
|
+
throw new UsageError(`plugin ${flags.plugin} has no skills`);
|
|
489
|
+
}
|
|
490
|
+
else if (append !== null && existing !== null) {
|
|
491
|
+
const covered = new Set(existing.cases.flatMap((item) => [...item.expect, ...item.expect_any, ...item.forbid, ...(item.first ? [item.first] : [])]));
|
|
492
|
+
targets = docs.filter((doc) => doc.kind === "skill" && doc.plugin === null && !covered.has(doc.name));
|
|
493
|
+
if (targets.length === 0) {
|
|
494
|
+
io.stderr.write(`nothing to draft: every skill already has cases in ${flags.append} (pass --skill to draft more)\n`);
|
|
495
|
+
return 0;
|
|
496
|
+
}
|
|
497
|
+
}
|
|
498
|
+
else {
|
|
499
|
+
targets = docs.filter((doc) => doc.kind === "skill" && doc.plugin === null);
|
|
500
|
+
}
|
|
501
|
+
if (targets.length === 0)
|
|
502
|
+
throw new UsageError("no skills found: put them in .claude/skills or pass --config-dir");
|
|
503
|
+
// before any model call: a refused file must not cost anything
|
|
504
|
+
const out = flags.out === undefined ? null : path.resolve(io.cwd, flags.out);
|
|
505
|
+
if (out !== null && fs.existsSync(out) && !flags.force)
|
|
506
|
+
throw new UsageError(`${flags.out} exists, use --force to overwrite`);
|
|
507
|
+
const adapter = pickAdapter(flags.agent ?? "claude", io, deps);
|
|
508
|
+
if (!adapter)
|
|
509
|
+
return 2;
|
|
510
|
+
if (!adapter.runBatch)
|
|
511
|
+
throw new UsageError(`agent ${adapter.name} cannot generate cases`);
|
|
512
|
+
const groups = [];
|
|
513
|
+
for (let i = 0; i < targets.length; i += 8)
|
|
514
|
+
groups.push(targets.slice(i, i + 8));
|
|
515
|
+
const results = await runPool(groups, flags.jobs, async (group) => {
|
|
516
|
+
try {
|
|
517
|
+
const result = await adapter.runBatch({
|
|
518
|
+
prompt: buildGenPrompt(group, docs, flags.perSkill),
|
|
519
|
+
schema: GEN_SCHEMA,
|
|
520
|
+
model: flags.model,
|
|
521
|
+
timeoutMs: flags.timeoutSec * 1000,
|
|
522
|
+
configDir: flags.configDir,
|
|
523
|
+
});
|
|
524
|
+
if (result.error) {
|
|
525
|
+
io.stderr.write(`skillcheck: gen failed for ${group.map((doc) => doc.name).join(", ")}: ${result.error}\n`);
|
|
526
|
+
return { cases: [], dropped: 0, costUsd: result.costUsd, failed: true };
|
|
527
|
+
}
|
|
528
|
+
const parsed = parseGenAnswer(result.structured, new Set(docs.map((doc) => doc.name)), new Set(group.map((doc) => doc.name)));
|
|
529
|
+
return { ...parsed, costUsd: result.costUsd, failed: false };
|
|
530
|
+
}
|
|
531
|
+
catch (e) {
|
|
532
|
+
const error = e.message;
|
|
533
|
+
io.stderr.write(`skillcheck: gen failed for ${group.map((doc) => doc.name).join(", ")}: ${error}\n`);
|
|
534
|
+
return { cases: [], dropped: 0, costUsd: null, failed: true };
|
|
535
|
+
}
|
|
536
|
+
});
|
|
537
|
+
const successful = results.filter((result) => result !== undefined && !result.failed);
|
|
538
|
+
if (successful.length === 0)
|
|
539
|
+
return 2;
|
|
540
|
+
// each answer is deduplicated on its own; groups can still repeat each other
|
|
541
|
+
const seen = new Set();
|
|
542
|
+
const cases = successful.flatMap((result) => result.cases).filter((item) => {
|
|
543
|
+
const key = item.query.trim();
|
|
544
|
+
if (seen.has(key))
|
|
545
|
+
return false;
|
|
546
|
+
seen.add(key);
|
|
547
|
+
return true;
|
|
548
|
+
});
|
|
549
|
+
const dropped = successful.reduce((total, result) => total + result.dropped, 0);
|
|
550
|
+
if (dropped > 0)
|
|
551
|
+
io.stderr.write(`note: dropped ${dropped} proposed cases naming unknown skills or none of the requested ones\n`);
|
|
552
|
+
const cost = genCost(results.map((result) => result?.costUsd ?? null));
|
|
553
|
+
if (cases.length === 0) {
|
|
554
|
+
io.stderr.write(`skillcheck: gen got no usable cases from the model (cost $${cost})\n`);
|
|
555
|
+
return 2;
|
|
556
|
+
}
|
|
557
|
+
if (append !== null) {
|
|
558
|
+
const oldQueries = new Set(existing.cases.map((item) => item.query.trim()));
|
|
559
|
+
const unique = cases.filter((item) => !oldQueries.has(item.query.trim()));
|
|
560
|
+
const duplicates = cases.length - unique.length;
|
|
561
|
+
if (unique.length > 0)
|
|
562
|
+
appendCases(append, unique, targets.map((doc) => doc.name), flags.model ?? null);
|
|
563
|
+
const suffix = duplicates > 0 ? `, ${duplicates} duplicates skipped` : "";
|
|
564
|
+
io.stdout.write(`appended ${unique.length} cases to ${flags.append} (${targets.length} skills${suffix}, cost $${cost})\n`);
|
|
565
|
+
return 0;
|
|
566
|
+
}
|
|
567
|
+
const text = genSuite(cases, { model: flags.model ?? null, skills: targets.map((doc) => doc.name), json: out !== null && path.extname(out).toLowerCase() === ".json" });
|
|
568
|
+
const counts = `${cases.length} cases for ${targets.length} skills, cost $${cost}`;
|
|
569
|
+
if (out === null) {
|
|
570
|
+
io.stdout.write(text);
|
|
571
|
+
io.stderr.write(`drafted ${counts}\n`);
|
|
572
|
+
return 0;
|
|
573
|
+
}
|
|
574
|
+
try {
|
|
575
|
+
fs.writeFileSync(out, text);
|
|
576
|
+
}
|
|
577
|
+
catch (e) {
|
|
578
|
+
throw new UsageError(`cannot write ${flags.out}: ${e.message}`);
|
|
579
|
+
}
|
|
580
|
+
io.stdout.write(`wrote ${flags.out} (${counts})\n`);
|
|
581
|
+
return 0;
|
|
582
|
+
}
|
|
583
|
+
function genCost(costs) {
|
|
584
|
+
const known = costs.filter((cost) => cost !== null);
|
|
585
|
+
const unknown = costs.length - known.length;
|
|
586
|
+
return known.length === 0 ? "?" : known.reduce((total, value) => total + value, 0).toFixed(2) +
|
|
587
|
+
(unknown > 0 ? ` + ${unknown} call${unknown === 1 ? "" : "s"} of unknown cost` : "");
|
|
588
|
+
}
|
|
589
|
+
function appendCases(file, cases, skills, model) {
|
|
590
|
+
const ext = path.extname(file).toLowerCase();
|
|
591
|
+
const ordered = orderCases(cases, skills);
|
|
592
|
+
const text = fs.readFileSync(file, "utf8");
|
|
593
|
+
let nextText;
|
|
594
|
+
if (ext === ".json") {
|
|
595
|
+
const data = JSON.parse(text);
|
|
596
|
+
if (Array.isArray(data))
|
|
597
|
+
data.push(...ordered.map(caseEntry));
|
|
598
|
+
else
|
|
599
|
+
data.cases.push(...ordered.map(caseEntry));
|
|
600
|
+
nextText = JSON.stringify(data, null, 2) + "\n";
|
|
601
|
+
}
|
|
602
|
+
else {
|
|
603
|
+
const doc = parseDocument(text);
|
|
604
|
+
// a bare list of cases is the older format, still accepted
|
|
605
|
+
const sequence = (isSeq(doc.contents) ? doc.contents : doc.get("cases", true));
|
|
606
|
+
for (const [i, item] of ordered.entries()) {
|
|
607
|
+
const node = doc.createNode(caseEntry(item));
|
|
608
|
+
// [name] like the rest of a cases file, not a block list
|
|
609
|
+
for (const key of ["expect", "forbid"]) {
|
|
610
|
+
const list = node.get(key, true);
|
|
611
|
+
if (isSeq(list))
|
|
612
|
+
list.flow = true;
|
|
613
|
+
}
|
|
614
|
+
if (i === 0)
|
|
615
|
+
node.commentBefore = ` gen draft (model: ${model ?? "default"}): review these cases`;
|
|
616
|
+
sequence.add(node);
|
|
617
|
+
}
|
|
618
|
+
nextText = doc.toString({ flowCollectionPadding: false });
|
|
619
|
+
}
|
|
620
|
+
parseSuite(ext === ".json" ? JSON.parse(nextText) : parseYaml(nextText));
|
|
621
|
+
fs.writeFileSync(file, nextText);
|
|
622
|
+
}
|
|
379
623
|
const UNCOVERED_WIDTH = 78;
|
|
380
624
|
function printLint(out, report, docs, suite) {
|
|
381
625
|
const casesPart = suite ? `${report.cases} cases` : "no cases file";
|
|
@@ -451,6 +695,15 @@ async function runCommand(file, flags, io, deps) {
|
|
|
451
695
|
const suite = await loadCases(file, io);
|
|
452
696
|
if (!suite)
|
|
453
697
|
return 2;
|
|
698
|
+
let baseline;
|
|
699
|
+
if (flags.baseline !== undefined) {
|
|
700
|
+
try {
|
|
701
|
+
baseline = parseBaseline(fs.readFileSync(path.resolve(io.cwd, flags.baseline), "utf8"));
|
|
702
|
+
}
|
|
703
|
+
catch (e) {
|
|
704
|
+
throw new UsageError(`cannot read baseline ${flags.baseline}: ${e.message}`);
|
|
705
|
+
}
|
|
706
|
+
}
|
|
454
707
|
const agent = flags.agent ?? suite.agent;
|
|
455
708
|
const adapter = pickAdapter(agent, io, deps);
|
|
456
709
|
if (!adapter)
|
|
@@ -477,8 +730,8 @@ async function runCommand(file, flags, io, deps) {
|
|
|
477
730
|
const startedAtMs = Date.now();
|
|
478
731
|
const fatal = new FatalStop();
|
|
479
732
|
const outcome = flags.batch
|
|
480
|
-
? await runBatched(adapter.runBatch.bind(adapter), items, flags, model,
|
|
481
|
-
: await runIndividually(adapter, items, flags, model, directive,
|
|
733
|
+
? await runBatched(adapter.runBatch.bind(adapter), items, flags, model, reporter, budget, fatal)
|
|
734
|
+
: await runIndividually(adapter, items, flags, model, directive, reporter, budget, fatal);
|
|
482
735
|
if (fatal.message !== null) {
|
|
483
736
|
// an environment problem, not a routing result: no summary, no reports
|
|
484
737
|
io.stderr.write(`skillcheck: stopped, the agent cannot run: ${fatal.message}\n`);
|
|
@@ -489,16 +742,41 @@ async function runCommand(file, flags, io, deps) {
|
|
|
489
742
|
for (const item of skipped)
|
|
490
743
|
reporter.caseSkipped(item.c);
|
|
491
744
|
const notStartedRuns = skipped.reduce((n, item) => n + item.runs.length - item.done, 0);
|
|
745
|
+
const skippedRunCosts = skipped.flatMap((item) => item.runs.flatMap((v) => (v ? [v.costUsd] : [])));
|
|
492
746
|
const ordered = outcome.results.filter((r) => r !== undefined);
|
|
493
747
|
const expected = expectedNames(selected);
|
|
494
748
|
const unavailable = outcome.sawAvailability ? expected.filter((name) => !outcome.available.has(name)) : [];
|
|
495
749
|
const pairs = confusion(ordered);
|
|
750
|
+
const report = buildReport({
|
|
751
|
+
file,
|
|
752
|
+
agent,
|
|
753
|
+
model: model ?? null,
|
|
754
|
+
startedAtMs,
|
|
755
|
+
durationMs: Date.now() - startedAtMs,
|
|
756
|
+
cases: items.map((item) => ({ c: item.c, threshold: item.threshold, result: outcome.results[item.jobIndex] ?? null })),
|
|
757
|
+
unavailable,
|
|
758
|
+
confusion: pairs,
|
|
759
|
+
batch: flags.batch,
|
|
760
|
+
estimatedCostUsd: budget.spent,
|
|
761
|
+
skippedRunCosts,
|
|
762
|
+
budgetUsd: flags.budget ?? null,
|
|
763
|
+
budgetReached: skipped.length > 0,
|
|
764
|
+
baseline,
|
|
765
|
+
baselineFile: flags.baseline,
|
|
766
|
+
suiteCases: suite.cases.map((c) => ({ id: c.id ?? null, query: c.query })),
|
|
767
|
+
});
|
|
496
768
|
reporter.summary(ordered, {
|
|
497
769
|
unavailable,
|
|
498
770
|
confusion: pairs,
|
|
499
771
|
skipped: skipped.length,
|
|
500
772
|
budget: skipped.length > 0 ? { limitUsd: flags.budget, spent: budget.spent, notStartedRuns } : undefined,
|
|
501
773
|
estimatedUsd: budget.spent,
|
|
774
|
+
skippedRunCosts,
|
|
775
|
+
baseline: report.baseline ? {
|
|
776
|
+
summary: report.baseline,
|
|
777
|
+
regressed: report.cases.filter((c) => c.change === "regressed").map((c) => `#${c.id ?? c.index}`),
|
|
778
|
+
fixed: report.cases.filter((c) => c.change === "fixed").map((c) => `#${c.id ?? c.index}`),
|
|
779
|
+
} : undefined,
|
|
502
780
|
});
|
|
503
781
|
if (flags.batch) {
|
|
504
782
|
// cases that only errored have no answer to confirm
|
|
@@ -508,20 +786,12 @@ async function runCommand(file, flags, io, deps) {
|
|
|
508
786
|
const hint = failed.length > 0 ? ` (--only ${failed.join(",")})` : "";
|
|
509
787
|
reporter.note(`batch mode: answers are the model's stated choice, not an actual Skill call — confirm failures with a normal run${hint}`);
|
|
510
788
|
}
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
cases: items.map((item) => ({ c: item.c, threshold: item.threshold, result: outcome.results[item.jobIndex] ?? null })),
|
|
518
|
-
unavailable,
|
|
519
|
-
confusion: pairs,
|
|
520
|
-
batch: flags.batch,
|
|
521
|
-
estimatedCostUsd: budget.spent,
|
|
522
|
-
budgetUsd: flags.budget ?? null,
|
|
523
|
-
budgetReached: skipped.length > 0,
|
|
524
|
-
});
|
|
789
|
+
if (baseline && baseline.model !== null && report.model !== null && baseline.model !== report.model) {
|
|
790
|
+
reporter.note(`baseline was run with model ${baseline.model}, this run with ${report.model}`);
|
|
791
|
+
}
|
|
792
|
+
if (baseline && baseline.batch !== report.batch) {
|
|
793
|
+
reporter.note(baseline.batch ? "baseline was a batch run, this one is not" : "this is a batch run, the baseline was not");
|
|
794
|
+
}
|
|
525
795
|
if (flags.json !== undefined) {
|
|
526
796
|
const payload = JSON.stringify(report, null, 2) + "\n";
|
|
527
797
|
if (flags.json === "-")
|
|
@@ -533,6 +803,9 @@ async function runCommand(file, flags, io, deps) {
|
|
|
533
803
|
writeReportFile(flags.junit, toJunit(report), "--junit");
|
|
534
804
|
if (flags.markdown !== undefined)
|
|
535
805
|
writeReportFile(flags.markdown, toMarkdown(report), "--markdown");
|
|
806
|
+
if (flags.onlyNewFailures) {
|
|
807
|
+
return report.cases.some((c) => c.status === "skipped" || (c.status === "failed" && (c.change === "regressed" || c.change === "new"))) ? 1 : 0;
|
|
808
|
+
}
|
|
536
809
|
return ordered.some((r) => !r.ok) || skipped.length > 0 ? 1 : 0;
|
|
537
810
|
}
|
|
538
811
|
function writeReportFile(path, text, flag) {
|
|
@@ -544,9 +817,9 @@ function writeReportFile(path, text, flag) {
|
|
|
544
817
|
}
|
|
545
818
|
}
|
|
546
819
|
/** One agent call per case: the normal mode. */
|
|
547
|
-
async function runIndividually(adapter, items, flags, model, directive,
|
|
820
|
+
async function runIndividually(adapter, items, flags, model, directive, reporter, budget, fatal) {
|
|
548
821
|
const totalRuns = items.reduce((n, item) => n + item.runs.length, 0);
|
|
549
|
-
reporter.header(items.length,
|
|
822
|
+
reporter.header(items.length, repeatLabel(items), 1, totalRuns);
|
|
550
823
|
const jobs = items.flatMap((item, jobIndex) => item.runs.map((_, runIndex) => ({ item, runIndex, jobIndex })));
|
|
551
824
|
// keep cases in file order even though runs of different cases interleave
|
|
552
825
|
const results = new Array(items.length);
|
|
@@ -582,17 +855,19 @@ async function runIndividually(adapter, items, flags, model, directive, repeat,
|
|
|
582
855
|
}
|
|
583
856
|
/** One agent call per chunk of cases; the model states its choice per request
|
|
584
857
|
* instead of loading skills. Verdicts flow through the same judge. */
|
|
585
|
-
async function runBatched(runBatch, items, flags, model,
|
|
858
|
+
async function runBatched(runBatch, items, flags, model, reporter, budget, fatal) {
|
|
586
859
|
const chunks = [];
|
|
587
860
|
for (let i = 0; i < items.length; i += flags.batchSize)
|
|
588
861
|
chunks.push(items.slice(i, i + flags.batchSize));
|
|
589
862
|
const rounds = Math.max(...items.map((item) => item.runs.length));
|
|
590
863
|
const jobs = [];
|
|
591
864
|
for (let round = 0; round < rounds; round++) {
|
|
865
|
+
// a round past every repeat in the chunk makes no call
|
|
592
866
|
for (const chunk of chunks)
|
|
593
|
-
|
|
867
|
+
if (chunk.some((item) => round < item.runs.length))
|
|
868
|
+
jobs.push({ chunk, round });
|
|
594
869
|
}
|
|
595
|
-
reporter.batchHeader(items.length,
|
|
870
|
+
reporter.batchHeader(items.length, repeatLabel(items), jobs.length);
|
|
596
871
|
const results = new Array(items.length);
|
|
597
872
|
await runPool(jobs, flags.jobs, async (job) => {
|
|
598
873
|
// a case with a smaller repeat only takes its first rounds
|
|
@@ -604,6 +879,8 @@ async function runBatched(runBatch, items, flags, model, repeat, reporter, budge
|
|
|
604
879
|
const parsed = parseBatchAnswer(br.structured, br.text);
|
|
605
880
|
const chunkError = br.error ?? parsed.error; // a chunk error is every case's error
|
|
606
881
|
fatal.note(br.error);
|
|
882
|
+
// one call, one budget entry: its usage prices it when the cost never came
|
|
883
|
+
budget.add(br.costUsd, br.usage ?? null);
|
|
607
884
|
active.forEach((item, i) => {
|
|
608
885
|
const answer = parsed.answers.get(i + 1);
|
|
609
886
|
const r = {
|
|
@@ -616,7 +893,6 @@ async function runBatched(runBatch, items, flags, model, repeat, reporter, budge
|
|
|
616
893
|
durationMs: br.durationMs,
|
|
617
894
|
};
|
|
618
895
|
const verdict = judge(item.c, r);
|
|
619
|
-
budget.add(verdict.costUsd);
|
|
620
896
|
item.runs[job.round] = verdict;
|
|
621
897
|
item.done++;
|
|
622
898
|
if (item.done === item.runs.length) {
|
|
@@ -628,6 +904,13 @@ async function runBatched(runBatch, items, flags, model, repeat, reporter, budge
|
|
|
628
904
|
}, { shouldStart: () => fatal.canStart(budget) });
|
|
629
905
|
return { results, available: new Set(), sawAvailability: false };
|
|
630
906
|
}
|
|
907
|
+
/** "3", or "1–3" when cases override the repeat. */
|
|
908
|
+
function repeatLabel(items) {
|
|
909
|
+
const counts = items.map((item) => item.runs.length);
|
|
910
|
+
const min = Math.min(...counts);
|
|
911
|
+
const max = Math.max(...counts);
|
|
912
|
+
return min === max ? String(min) : `${min}–${max}`;
|
|
913
|
+
}
|
|
631
914
|
function selectCases(suite, agent, only) {
|
|
632
915
|
const forAgent = suite.cases.filter((c) => !c.agents || c.agents.includes(agent));
|
|
633
916
|
if (!only)
|
package/dist/gen.js
ADDED
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
import { parse as parseYaml } from "yaml";
|
|
2
|
+
import { parseSuite } from "./cases.js";
|
|
3
|
+
import { yamlName } from "./import.js";
|
|
4
|
+
export const GEN_SCHEMA = {
|
|
5
|
+
type: "object",
|
|
6
|
+
properties: {
|
|
7
|
+
cases: {
|
|
8
|
+
type: "array",
|
|
9
|
+
items: {
|
|
10
|
+
type: "object",
|
|
11
|
+
properties: {
|
|
12
|
+
query: { type: "string" },
|
|
13
|
+
skill: { type: ["string", "null"] },
|
|
14
|
+
avoid: { type: ["string", "null"] },
|
|
15
|
+
},
|
|
16
|
+
required: ["query", "skill", "avoid"],
|
|
17
|
+
},
|
|
18
|
+
},
|
|
19
|
+
},
|
|
20
|
+
required: ["cases"],
|
|
21
|
+
};
|
|
22
|
+
/** Order generated cases by their requested skills and convert one to a suite entry. */
|
|
23
|
+
export function orderCases(cases, skills) {
|
|
24
|
+
const buckets = new Map(skills.map((skill) => [skill, []]));
|
|
25
|
+
const rest = [];
|
|
26
|
+
for (const item of cases) {
|
|
27
|
+
const target = item.avoid ?? item.skill;
|
|
28
|
+
const bucket = target === null ? undefined : buckets.get(target);
|
|
29
|
+
if (bucket)
|
|
30
|
+
bucket.push(item);
|
|
31
|
+
else
|
|
32
|
+
rest.push(item);
|
|
33
|
+
}
|
|
34
|
+
return [...skills.flatMap((skill) => buckets.get(skill) ?? []), ...rest];
|
|
35
|
+
}
|
|
36
|
+
/** A near miss owned by a neighbour gets both: expect the neighbour, forbid the target. */
|
|
37
|
+
export function caseEntry(item) {
|
|
38
|
+
return {
|
|
39
|
+
query: item.query,
|
|
40
|
+
...(item.skill !== null ? { expect: [item.skill] } : {}),
|
|
41
|
+
...(item.avoid !== null ? { forbid: [item.avoid] } : {}),
|
|
42
|
+
};
|
|
43
|
+
}
|
|
44
|
+
/** Build the case-writing prompt from installed descriptions. */
|
|
45
|
+
export function buildGenPrompt(targets, context, perSkill) {
|
|
46
|
+
const contextLines = context.map((doc) => `- ${doc.name}: ${JSON.stringify(doc.description.replace(/\s+/g, " ").slice(0, 300))}`);
|
|
47
|
+
const near = Math.ceil(perSkill / 2);
|
|
48
|
+
const targetLines = targets.map((doc) => `For ${doc.name}: write ${perSkill} positive requests and ${near} near misses. ` +
|
|
49
|
+
`Positive requests have skill ${JSON.stringify(doc.name)} and avoid null. ` +
|
|
50
|
+
`Near misses have avoid ${JSON.stringify(doc.name)}, and skill is another listed skill or null.`);
|
|
51
|
+
return [
|
|
52
|
+
"This is a test-writing task. Do not load any skill, do not use tools other than the structured output.",
|
|
53
|
+
"Context skills:",
|
|
54
|
+
...contextLines,
|
|
55
|
+
"For each target skill, write realistic first messages a user would send. They should vary in wording and length, must not name the skill, and should be written in the language of the skill description.",
|
|
56
|
+
...targetLines,
|
|
57
|
+
"Return only the structured output. Each request must be a JSON string.",
|
|
58
|
+
].join("\n");
|
|
59
|
+
}
|
|
60
|
+
// targets: the skills asked for; every case must load or avoid one of them
|
|
61
|
+
export function parseGenAnswer(structured, known, targets = known) {
|
|
62
|
+
if (typeof structured !== "object" || structured === null || Array.isArray(structured))
|
|
63
|
+
return { cases: [], dropped: 0 };
|
|
64
|
+
const rawCases = structured.cases;
|
|
65
|
+
if (!Array.isArray(rawCases))
|
|
66
|
+
return { cases: [], dropped: 0 };
|
|
67
|
+
const cases = [];
|
|
68
|
+
const seen = new Set();
|
|
69
|
+
let dropped = 0;
|
|
70
|
+
for (const raw of rawCases) {
|
|
71
|
+
if (typeof raw !== "object" || raw === null || Array.isArray(raw)) {
|
|
72
|
+
dropped++;
|
|
73
|
+
continue;
|
|
74
|
+
}
|
|
75
|
+
const item = raw;
|
|
76
|
+
const query = item.query;
|
|
77
|
+
const skill = item.skill === null ? null : item.skill;
|
|
78
|
+
const avoid = item.avoid === null ? null : item.avoid;
|
|
79
|
+
if (typeof query !== "string" || query.trim() === "" ||
|
|
80
|
+
(skill !== null && (typeof skill !== "string" || !known.has(skill))) ||
|
|
81
|
+
(avoid !== null && (typeof avoid !== "string" || !known.has(avoid))) ||
|
|
82
|
+
(skill === null && avoid === null) || skill === avoid || seen.has(query) ||
|
|
83
|
+
!((skill !== null && targets.has(skill)) || (avoid !== null && targets.has(avoid)))) {
|
|
84
|
+
dropped++;
|
|
85
|
+
continue;
|
|
86
|
+
}
|
|
87
|
+
seen.add(query);
|
|
88
|
+
cases.push({ query, skill: skill, avoid: avoid });
|
|
89
|
+
}
|
|
90
|
+
return { cases, dropped };
|
|
91
|
+
}
|
|
92
|
+
export function genSuite(cases, meta) {
|
|
93
|
+
const ordered = orderCases(cases, meta.skills);
|
|
94
|
+
if (meta.json) {
|
|
95
|
+
// JSON has no comments: the review note lives only in the command output
|
|
96
|
+
const data = { agent: "claude", repeat: 1, threshold: 1, cases: ordered.map(caseEntry) };
|
|
97
|
+
parseSuite(data);
|
|
98
|
+
return JSON.stringify(data, null, 2) + "\n";
|
|
99
|
+
}
|
|
100
|
+
const lines = [
|
|
101
|
+
`# Draft cases written by skillcheck gen (model: ${meta.model ?? "default"}) for: ${meta.skills.join(", ")}.`,
|
|
102
|
+
"# Review every case: the model guessed what should route where. Delete what is wrong,",
|
|
103
|
+
"# keep what matches how people really ask, then `skillcheck run --batch`.",
|
|
104
|
+
"agent: claude",
|
|
105
|
+
"repeat: 1",
|
|
106
|
+
"threshold: 1.0",
|
|
107
|
+
"cases:",
|
|
108
|
+
];
|
|
109
|
+
for (const item of ordered) {
|
|
110
|
+
lines.push(` - query: ${JSON.stringify(item.query)}`);
|
|
111
|
+
if (item.skill !== null)
|
|
112
|
+
lines.push(` expect: [${yamlName(item.skill)}]`);
|
|
113
|
+
if (item.avoid !== null)
|
|
114
|
+
lines.push(` forbid: [${yamlName(item.avoid)}]`);
|
|
115
|
+
}
|
|
116
|
+
const text = lines.join("\n") + "\n";
|
|
117
|
+
parseSuite(parseYaml(text));
|
|
118
|
+
return text;
|
|
119
|
+
}
|
package/dist/import.js
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
import * as path from "node:path";
|
|
2
|
+
import { parse as parseYaml } from "yaml";
|
|
3
|
+
import { ConfigError, parseSuite } from "./cases.js";
|
|
4
|
+
const PLAIN_NAME = /^[A-Za-z0-9._:/-]+$/;
|
|
5
|
+
export function yamlName(name) {
|
|
6
|
+
return PLAIN_NAME.test(name) && parseYaml(name) === name ? name : JSON.stringify(name);
|
|
7
|
+
}
|
|
8
|
+
/**
|
|
9
|
+
* A skill-creator trigger eval set ([{query, should_trigger}], one skill) as a
|
|
10
|
+
* cases file: should_trigger true becomes expect, false becomes forbid.
|
|
11
|
+
*/
|
|
12
|
+
export function evalSetToSuite(data, skill, source) {
|
|
13
|
+
const errors = [];
|
|
14
|
+
if (!Array.isArray(data) || data.length === 0) {
|
|
15
|
+
throw new ConfigError(["expected a JSON array of {query, should_trigger}"]);
|
|
16
|
+
}
|
|
17
|
+
const items = [];
|
|
18
|
+
data.forEach((item, i) => {
|
|
19
|
+
const n = i + 1;
|
|
20
|
+
if (typeof item !== "object" || item === null || Array.isArray(item)) {
|
|
21
|
+
errors.push(`item ${n}: expected an object with query and should_trigger`);
|
|
22
|
+
return;
|
|
23
|
+
}
|
|
24
|
+
const { query, should_trigger: trigger } = item;
|
|
25
|
+
if (typeof query !== "string" || query.trim() === "")
|
|
26
|
+
errors.push(`item ${n}: query must be a non-empty string`);
|
|
27
|
+
if (typeof trigger !== "boolean")
|
|
28
|
+
errors.push(`item ${n}: should_trigger must be true or false`);
|
|
29
|
+
if (typeof query === "string" && typeof trigger === "boolean")
|
|
30
|
+
items.push({ query, trigger });
|
|
31
|
+
});
|
|
32
|
+
if (errors.length > 0)
|
|
33
|
+
throw new ConfigError(errors);
|
|
34
|
+
// plain only when YAML reads it back as the same string: not true, null, 123
|
|
35
|
+
const name = yamlName(skill);
|
|
36
|
+
const lines = [
|
|
37
|
+
`# skillcheck cases imported from ${path.basename(source)} (skill-creator trigger eval set).`,
|
|
38
|
+
"# should_trigger: true became expect, false became forbid: another skill may",
|
|
39
|
+
`# still load there, only ${skill} must not.`,
|
|
40
|
+
"agent: claude",
|
|
41
|
+
"repeat: 1",
|
|
42
|
+
"threshold: 1.0",
|
|
43
|
+
"cases:",
|
|
44
|
+
];
|
|
45
|
+
for (const { query, trigger } of items) {
|
|
46
|
+
// a JSON string is a valid YAML double-quoted scalar, newlines and quotes included
|
|
47
|
+
lines.push(` - query: ${JSON.stringify(query)}`, ` ${trigger ? "expect" : "forbid"}: [${name}]`);
|
|
48
|
+
}
|
|
49
|
+
const text = lines.join("\n") + "\n";
|
|
50
|
+
parseSuite(parseYaml(text));
|
|
51
|
+
const positive = items.filter((i) => i.trigger).length;
|
|
52
|
+
return { text, positive, negative: items.length - positive };
|
|
53
|
+
}
|
package/dist/junit.js
CHANGED
|
@@ -19,6 +19,12 @@ export function toJunit(report) {
|
|
|
19
19
|
properties.push(` <property name="unknownCostRuns" value="${report.summary.unknownCostRuns}"/>`);
|
|
20
20
|
properties.push(` <property name="estimatedCostUsd" value="${report.summary.estimatedCostUsd.toFixed(2)}"/>`);
|
|
21
21
|
}
|
|
22
|
+
if (report.baseline) {
|
|
23
|
+
properties.push(` <property name="baselineRegressed" value="${report.baseline.regressed}"/>`);
|
|
24
|
+
properties.push(` <property name="baselineFixed" value="${report.baseline.fixed}"/>`);
|
|
25
|
+
properties.push(` <property name="baselineNew" value="${report.baseline.new}"/>`);
|
|
26
|
+
properties.push(` <property name="baselineRemoved" value="${report.baseline.removed}"/>`);
|
|
27
|
+
}
|
|
22
28
|
properties.push(" </properties>");
|
|
23
29
|
const cases = rows.map(({ c, kind }) => testcase(c, kind, suiteName));
|
|
24
30
|
return [
|
package/dist/markdown.js
CHANGED
|
@@ -6,6 +6,10 @@ export function toMarkdown(report) {
|
|
|
6
6
|
const lines = [MARKDOWN_MARKER, header(report)];
|
|
7
7
|
lines.push("");
|
|
8
8
|
lines.push(metaLine(report));
|
|
9
|
+
if (report.baseline) {
|
|
10
|
+
lines.push("");
|
|
11
|
+
lines.push(`> Since baseline: ${baselineText(report.baseline)}`);
|
|
12
|
+
}
|
|
9
13
|
if (report.batch) {
|
|
10
14
|
lines.push("");
|
|
11
15
|
lines.push("> Batch mode: answers are the model's stated choice, not an actual Skill call. Confirm failures with a normal run.");
|
|
@@ -96,21 +100,36 @@ function failMark(c) {
|
|
|
96
100
|
return bad.length > 0 && bad.every((r) => r.error !== null) ? "⚠️" : "❌";
|
|
97
101
|
}
|
|
98
102
|
/** The first non-ok run's reason, the pass share prefixed over several runs,
|
|
99
|
-
* a diagnosis appended in italics. */
|
|
103
|
+
* a diagnosis appended in italics, then the case note. */
|
|
100
104
|
function reasonCell(c) {
|
|
101
105
|
const bad = c.runs.filter((r) => !r.ok);
|
|
102
106
|
const parts = c.runs.length > 1 ? [`${c.passed}/${c.runs.length}`, bad[0]?.reason ?? ""] : [bad[0]?.reason ?? ""];
|
|
103
107
|
let text = parts.join(" · ");
|
|
108
|
+
if (c.change === "regressed")
|
|
109
|
+
text = `**regressed** · ${text}`;
|
|
110
|
+
else if (c.change === "new")
|
|
111
|
+
text = `new · ${text}`;
|
|
104
112
|
const diagnosis = bad.find((r) => r.diagnosis !== null)?.diagnosis;
|
|
105
113
|
if (diagnosis)
|
|
106
114
|
text += ` — _${diagnosis}_`;
|
|
115
|
+
if (c.note !== null)
|
|
116
|
+
text += ` · note: ${c.note}`;
|
|
107
117
|
return text;
|
|
108
118
|
}
|
|
109
119
|
function passedRow(c) {
|
|
110
120
|
const score = c.runs.length > 1 ? ` (${c.passed}/${c.runs.length})` : "";
|
|
111
|
-
const cells = [`${caseLabel(c)}${score}`, oneLine(c.query), loaded(c)];
|
|
121
|
+
const cells = [`${caseLabel(c)}${score}`, oneLine(c.query), `${loaded(c)}${c.change === "fixed" ? " · fixed" : ""}`];
|
|
112
122
|
return `| ${cells.map(esc).join(" | ")} |`;
|
|
113
123
|
}
|
|
124
|
+
function baselineText(summary) {
|
|
125
|
+
const parts = [
|
|
126
|
+
summary.regressed > 0 ? `**${summary.regressed} regressed**` : null,
|
|
127
|
+
summary.fixed > 0 ? `${summary.fixed} fixed` : null,
|
|
128
|
+
summary.new > 0 ? `${summary.new} new` : null,
|
|
129
|
+
summary.removed > 0 ? `${summary.removed} removed` : null,
|
|
130
|
+
].filter((p) => p !== null);
|
|
131
|
+
return parts.length > 0 ? `${parts.join(", ")}.` : "no changes.";
|
|
132
|
+
}
|
|
114
133
|
function caseLabel(c) {
|
|
115
134
|
return `#${c.id ?? c.index}`;
|
|
116
135
|
}
|
package/dist/report.js
CHANGED
|
@@ -11,7 +11,7 @@ export class Reporter {
|
|
|
11
11
|
this.useColor = out.isTTY === true && !process.env.NO_COLOR;
|
|
12
12
|
}
|
|
13
13
|
header(nCases, repeat, nAgents, totalRuns) {
|
|
14
|
-
const runs = totalRuns ?? nCases * repeat * nAgents;
|
|
14
|
+
const runs = totalRuns ?? nCases * Number(repeat) * nAgents;
|
|
15
15
|
const agents = nAgents === 1 ? "agent" : "agents";
|
|
16
16
|
this.write(`${nCases} cases × ${repeat} repeat × ${nAgents} ${agents} = ${runs} runs\n`);
|
|
17
17
|
}
|
|
@@ -36,6 +36,8 @@ export class Reporter {
|
|
|
36
36
|
if (diagnosed?.diagnosis) {
|
|
37
37
|
this.write(` ${this.paint("33", `diagnosis ${label}: ${diagnosed.diagnosis}`)}\n`);
|
|
38
38
|
}
|
|
39
|
+
if (!res.ok && res.case.note)
|
|
40
|
+
this.write(` ${this.paint("2", `note ${label}: ${oneLine(res.case.note)}`)}\n`);
|
|
39
41
|
}
|
|
40
42
|
/** A case the budget never let start; printed after all real runs settle. */
|
|
41
43
|
caseSkipped(c) {
|
|
@@ -68,7 +70,18 @@ export class Reporter {
|
|
|
68
70
|
const failed = results.filter((r) => !r.ok).length;
|
|
69
71
|
const skipped = extra?.skipped ?? 0;
|
|
70
72
|
const skippedPart = skipped > 0 ? `, ${skipped} skipped` : "";
|
|
71
|
-
|
|
73
|
+
const costs = [...verdicts.map((v) => v.costUsd), ...(extra?.skippedRunCosts ?? [])];
|
|
74
|
+
this.write(`${failed} failed${skippedPart} of ${results.length + skipped} · runs ${costs.length} · ${costLine(costs, extra?.estimatedUsd)}\n`);
|
|
75
|
+
if (extra?.baseline) {
|
|
76
|
+
const b = extra.baseline;
|
|
77
|
+
const parts = [
|
|
78
|
+
b.summary.regressed > 0 ? this.paint("31", `${b.summary.regressed} regressed (${b.regressed.join(", ")})`) : null,
|
|
79
|
+
b.summary.fixed > 0 ? this.paint("32", `${b.summary.fixed} fixed (${b.fixed.join(", ")})`) : null,
|
|
80
|
+
b.summary.new > 0 ? `${b.summary.new} new` : null,
|
|
81
|
+
b.summary.removed > 0 ? `${b.summary.removed} removed` : null,
|
|
82
|
+
].filter((p) => p !== null);
|
|
83
|
+
this.write(`vs baseline: ${parts.length > 0 ? parts.join(", ") : "no changes"}\n`);
|
|
84
|
+
}
|
|
72
85
|
}
|
|
73
86
|
paint(code, text) {
|
|
74
87
|
return this.useColor ? `\u001b[${code}m${text}${RESET}` : text;
|
package/dist/results.js
CHANGED
|
@@ -1,9 +1,12 @@
|
|
|
1
|
+
import { compare } from "./baseline.js";
|
|
1
2
|
import { readVersion } from "./version.js";
|
|
2
3
|
/** The single source of truth for --json and --junit; computed once per run. */
|
|
3
4
|
export function buildReport(input) {
|
|
4
5
|
const verdicts = input.cases.flatMap((e) => e.result?.runs ?? []);
|
|
5
|
-
const costs = verdicts.map((v) => v.costUsd);
|
|
6
|
+
const costs = [...verdicts.map((v) => v.costUsd), ...(input.skippedRunCosts ?? [])];
|
|
6
7
|
const known = costs.filter((c) => c !== null);
|
|
8
|
+
const cases = input.cases.map(toCaseReport);
|
|
9
|
+
const comparison = input.baseline ? compare(cases, input.baseline, input.baselineFile ?? input.baseline.file, input.suiteCases ?? cases) : null;
|
|
7
10
|
return {
|
|
8
11
|
tool: "skillcheck",
|
|
9
12
|
version: readVersion(),
|
|
@@ -17,7 +20,7 @@ export function buildReport(input) {
|
|
|
17
20
|
cases: input.cases.length,
|
|
18
21
|
failed: input.cases.filter((e) => e.result && !e.result.ok).length,
|
|
19
22
|
skipped: input.cases.filter((e) => !e.result).length,
|
|
20
|
-
runs:
|
|
23
|
+
runs: costs.length,
|
|
21
24
|
costUsd: known.reduce((a, b) => a + b, 0),
|
|
22
25
|
unknownCostRuns: costs.length - known.length,
|
|
23
26
|
estimatedCostUsd: input.estimatedCostUsd,
|
|
@@ -27,7 +30,8 @@ export function buildReport(input) {
|
|
|
27
30
|
},
|
|
28
31
|
unavailable: input.unavailable,
|
|
29
32
|
confusion: input.confusion,
|
|
30
|
-
cases:
|
|
33
|
+
cases: comparison ? cases.map((c, i) => ({ ...c, change: comparison.changes[i] ?? null })) : cases,
|
|
34
|
+
baseline: comparison?.summary ?? null,
|
|
31
35
|
};
|
|
32
36
|
}
|
|
33
37
|
function toCaseReport(e) {
|
|
@@ -46,5 +50,6 @@ function toCaseReport(e) {
|
|
|
46
50
|
passed: e.result?.passed ?? 0,
|
|
47
51
|
threshold: e.result?.threshold ?? e.threshold,
|
|
48
52
|
runs: e.result?.runs ?? [],
|
|
53
|
+
change: null,
|
|
49
54
|
};
|
|
50
55
|
}
|