@icntswm/skillcheck 1.2.0 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +73 -92
- package/dist/baseline.js +59 -0
- package/dist/cli.js +237 -19
- package/dist/gen.js +119 -0
- package/dist/import.js +4 -1
- package/dist/junit.js +6 -0
- package/dist/markdown.js +18 -1
- package/dist/report.js +10 -0
- package/dist/results.js +6 -1
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -8,50 +8,49 @@
|
|
|
8
8
|
**Regression tests for Claude Code skills.**
|
|
9
9
|
|
|
10
10
|
You edited a skill description, installed a plugin or switched models, and now
|
|
11
|
-
some
|
|
11
|
+
some requests load the wrong skill. Nothing warns you: the agent still
|
|
12
12
|
answers, just with the wrong instructions. skillcheck catches this before your
|
|
13
13
|
users do.
|
|
14
14
|
|
|
15
15
|

|
|
16
16
|
|
|
17
|
-
A real run on the [demo](examples/demo): one
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
requests.
|
|
17
|
+
<sub>A real run on the [demo](examples/demo): one description got wider and
|
|
18
|
+
started taking its neighbour's requests. The free `lint` flags it, one batch
|
|
19
|
+
call finds both misrouted requests.</sub>
|
|
21
20
|
|
|
22
|
-
##
|
|
21
|
+
## Quick start
|
|
23
22
|
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
## How it works
|
|
23
|
+
Needs Node 20+ and [Claude Code](https://docs.anthropic.com/en/docs/claude-code)
|
|
24
|
+
(`claude`) in `PATH`, logged in.
|
|
25
|
+
|
|
26
|
+
```sh
|
|
27
|
+
npm install -g @icntswm/skillcheck
|
|
28
|
+
|
|
29
|
+
skillcheck init # skillcheck.yaml listing your skills; add real requests
|
|
30
|
+
skillcheck lint # free static checks, no model calls
|
|
31
|
+
skillcheck run --batch # cheap pre-check: one call per 25 cases
|
|
32
|
+
skillcheck run --only 2,5 # confirm what the batch flagged with real runs
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
| Command | What it does |
|
|
36
|
+
|---|---|
|
|
37
|
+
| `init` | starter `skillcheck.yaml` with your skill names, free |
|
|
38
|
+
| `gen` | the model drafts cases from your descriptions, for you to review |
|
|
39
|
+
| `import` | turns a skill-creator trigger eval set into cases |
|
|
40
|
+
| `check` | validates the file and catches misspelled or renamed skills, free |
|
|
41
|
+
| `lint` | short or look-alike descriptions, free |
|
|
42
|
+
| `list` | the skills and commands Claude Code sees, free |
|
|
43
|
+
| `run` | runs the cases; `--batch` for the cheap pre-check |
|
|
44
|
+
|
|
45
|
+
Every option, the default file lookup and exit codes are in
|
|
46
|
+
[Commands](docs/commands.md).
|
|
47
|
+
|
|
48
|
+
No cases yet? `skillcheck gen -o skillcheck.yaml` lets the model draft them from
|
|
49
|
+
your skill descriptions, for you to review. Coming from Anthropic's
|
|
50
|
+
skill-creator? `skillcheck import eval_set.json --skill <name>` converts its
|
|
51
|
+
trigger eval set. To try without installing: `npx @icntswm/skillcheck lint`.
|
|
52
|
+
|
|
53
|
+
## How it works
|
|
55
54
|
|
|
56
55
|
Write requests the way you actually type them, and say which skill must, or
|
|
57
56
|
must not, load:
|
|
@@ -66,84 +65,66 @@ cases:
|
|
|
66
65
|
```
|
|
67
66
|
|
|
68
67
|
skillcheck sends each request to a headless `claude -p`, watches which skills
|
|
69
|
-
the model loads, and stops the process as soon as the answer is clear.
|
|
68
|
+
the model loads, and stops the process as soon as the answer is clear. Runs go
|
|
69
|
+
in plan mode with edits, shell, web and MCP tools off, so your files are never
|
|
70
|
+
touched.
|
|
70
71
|
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
Start at the cheapest level and go up only when it finds nothing.
|
|
72
|
+
Three levels, from free to exact. Go up only when the cheaper one finds nothing:
|
|
74
73
|
|
|
75
74
|
| Command | Model calls | What it tells you |
|
|
76
75
|
|---|---|---|
|
|
77
|
-
| `skillcheck lint` | none | short
|
|
76
|
+
| `skillcheck lint` | none | short or look-alike descriptions, cases that share few words with their skill |
|
|
78
77
|
| `skillcheck run --batch` | one per 25 cases | which skill the model *says* it would load |
|
|
79
78
|
| `skillcheck run` | one per case | which skill the model *actually* loads |
|
|
80
79
|
|
|
81
|
-
On the demo
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
80
|
+
On the [demo](examples/demo) (six skills, twelve cases, a bad edit of two
|
|
81
|
+
descriptions) `lint` flagged the short description, and both `run --batch` and
|
|
82
|
+
`run` caught the two misrouted cases. One batch call cost $0.05–0.10 and agreed
|
|
83
|
+
with the full run in 34 checks out of 34.
|
|
85
84
|
|
|
86
|
-
|
|
87
|
-
edit of two descriptions.
|
|
88
|
-
|
|
89
|
-
| | good skills | bad edit |
|
|
90
|
-
|---|---|---|
|
|
91
|
-
| `lint` | no problems | flags the too-short description |
|
|
92
|
-
| `run --batch` | 12/12 passed | catches both misrouted cases |
|
|
93
|
-
| `run` | 5/5 passed | catches the same two cases |
|
|
94
|
-
|
|
95
|
-
The demo README also lists edits that did *not* break routing, and why.
|
|
96
|
-
|
|
97
|
-
## Install
|
|
98
|
-
|
|
99
|
-
Needs Node 20+ and [Claude Code](https://docs.anthropic.com/en/docs/claude-code)
|
|
100
|
-
(`claude`) in `PATH`, logged in.
|
|
101
|
-
|
|
102
|
-
```
|
|
103
|
-
npm install -g @icntswm/skillcheck
|
|
104
|
-
```
|
|
85
|
+
## What you get
|
|
105
86
|
|
|
106
|
-
The
|
|
107
|
-
|
|
87
|
+
- 🎯 **The real routing decision**, read from Claude Code itself, not guessed from keywords.
|
|
88
|
+
- 💸 **You pay for the decision, not the work**: a run stops the moment a skill is picked.
|
|
89
|
+
- 🔍 **Points at the fix**: the confusion block shows which skill took whose requests.
|
|
90
|
+
- 🧪 **Honest about randomness**: `--repeat` and `threshold` tell a flaky case from a broken one.
|
|
91
|
+
- 🏷️ **Catches renames**: case names are checked against the skills the agent really has.
|
|
92
|
+
- 📄 **Plain YAML, one dependency**: cases live next to the skills they test.
|
|
108
93
|
|
|
109
|
-
##
|
|
94
|
+
## In CI
|
|
110
95
|
|
|
96
|
+
```yaml
|
|
97
|
+
- uses: icntswm/skillcheck@v1
|
|
98
|
+
env:
|
|
99
|
+
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
|
100
|
+
with:
|
|
101
|
+
model: sonnet
|
|
102
|
+
budget: 2
|
|
111
103
|
```
|
|
112
|
-
skillcheck init # writes skillcheck.yaml listing your skills
|
|
113
|
-
# then add a few real requests per skill
|
|
114
|
-
skillcheck check # validates the file, no model calls
|
|
115
|
-
skillcheck lint # free static checks
|
|
116
|
-
skillcheck run --batch # cheap pre-check, one call
|
|
117
|
-
skillcheck run --only 2,5 # confirm what the batch flagged
|
|
118
|
-
```
|
|
119
|
-
|
|
120
|
-
`skillcheck list` prints the skills and slash commands Claude Code sees. It
|
|
121
|
-
stops Claude Code before the first model call, so it costs nothing.
|
|
122
104
|
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
105
|
+
The action tests only the skills in your repository, comments the report on the
|
|
106
|
+
pull request and writes JUnit, JSON and Markdown reports. `--budget` caps the
|
|
107
|
+
spend; `--baseline` with `--only-new-failures` compares with `main` and fails
|
|
108
|
+
only on new failures. Run it after you rewrite a description, add a skill that
|
|
109
|
+
sounds like an existing one, install a plugin, or switch models.
|
|
126
110
|
|
|
127
111
|
## Documentation
|
|
128
112
|
|
|
129
113
|
| | |
|
|
130
114
|
|---|---|
|
|
131
|
-
| [
|
|
115
|
+
| [Commands](docs/commands.md) | every command and option, exit codes |
|
|
116
|
+
| [Writing cases](docs/writing-cases.md) | the file format, what makes a good case, `gen` and `import` |
|
|
132
117
|
| [Cost](docs/cost.md) | what a run costs and how to spend less |
|
|
133
|
-
| [Reports and CI](docs/ci.md) | confusion block,
|
|
118
|
+
| [Reports and CI](docs/ci.md) | confusion block, reports, GitHub Actions, comparing with `main` |
|
|
134
119
|
| [How it works](docs/how-it-works.md) | what happens inside a run, and the limits of each level |
|
|
135
120
|
|
|
136
|
-
`skillcheck --help` lists every command and option.
|
|
137
|
-
|
|
138
121
|
## Status
|
|
139
122
|
|
|
140
|
-
Works with Claude Code
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
languages that put spaces between words, and is of little use for Chinese,
|
|
146
|
-
Japanese or Korean.
|
|
123
|
+
Works with Claude Code; other agents with skills can be added behind a small
|
|
124
|
+
adapter interface. Requests can be in any language: `run` asks the model
|
|
125
|
+
itself. `lint` compares words, so it is tuned for English and Russian, works
|
|
126
|
+
roughly for other languages with spaces between words, and is of little use for
|
|
127
|
+
Chinese, Japanese or Korean.
|
|
147
128
|
|
|
148
129
|
## License
|
|
149
130
|
|
package/dist/baseline.js
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
const STATUSES = ["passed", "failed", "skipped"];
|
|
2
|
+
/** Parse a --json report; checked up front so a bad file fails before any model call. */
|
|
3
|
+
export function parseBaseline(text) {
|
|
4
|
+
const value = JSON.parse(text);
|
|
5
|
+
if (value === null || typeof value !== "object" || value.tool !== "skillcheck" || !Array.isArray(value.cases)) {
|
|
6
|
+
throw new Error("not a skillcheck --json report");
|
|
7
|
+
}
|
|
8
|
+
value.cases.forEach((c, i) => {
|
|
9
|
+
const entry = c;
|
|
10
|
+
if (typeof entry !== "object" || entry === null || typeof entry.query !== "string" || !STATUSES.includes(entry.status) ||
|
|
11
|
+
(entry.id !== null && typeof entry.id !== "string")) {
|
|
12
|
+
throw new Error(`case ${i + 1} is not a report case (needs query, id and status passed, failed or skipped)`);
|
|
13
|
+
}
|
|
14
|
+
});
|
|
15
|
+
return value;
|
|
16
|
+
}
|
|
17
|
+
/**
|
|
18
|
+
* Compare current cases with a previous report. suite is every case of the cases file, so cases
|
|
19
|
+
* left out by --only or --skill do not count as removed.
|
|
20
|
+
*/
|
|
21
|
+
export function compare(current, baseline, file, suite = current) {
|
|
22
|
+
const previous = new Map();
|
|
23
|
+
for (const c of baseline.cases) {
|
|
24
|
+
const key = caseKey(c);
|
|
25
|
+
if (!previous.has(key))
|
|
26
|
+
previous.set(key, c);
|
|
27
|
+
}
|
|
28
|
+
let regressed = 0;
|
|
29
|
+
let fixed = 0;
|
|
30
|
+
let newCount = 0;
|
|
31
|
+
const changes = current.map((c) => {
|
|
32
|
+
const old = previous.get(caseKey(c));
|
|
33
|
+
if (c.status === "skipped")
|
|
34
|
+
return null;
|
|
35
|
+
// a case the baseline budget skipped has no result to compare with
|
|
36
|
+
if (old === undefined || old.status === "skipped") {
|
|
37
|
+
newCount++;
|
|
38
|
+
return "new";
|
|
39
|
+
}
|
|
40
|
+
if (old.status === "passed" && c.status === "failed") {
|
|
41
|
+
regressed++;
|
|
42
|
+
return "regressed";
|
|
43
|
+
}
|
|
44
|
+
if (old.status === "failed" && c.status === "passed") {
|
|
45
|
+
fixed++;
|
|
46
|
+
return "fixed";
|
|
47
|
+
}
|
|
48
|
+
return null;
|
|
49
|
+
});
|
|
50
|
+
const suiteKeys = new Set(suite.map(caseKey));
|
|
51
|
+
let removed = 0;
|
|
52
|
+
for (const key of previous.keys())
|
|
53
|
+
if (!suiteKeys.has(key))
|
|
54
|
+
removed++;
|
|
55
|
+
return { changes, summary: { file, regressed, fixed, new: newCount, removed } };
|
|
56
|
+
}
|
|
57
|
+
function caseKey(c) {
|
|
58
|
+
return c.id !== null ? `id:${c.id}` : `query:${c.query}`;
|
|
59
|
+
}
|
package/dist/cli.js
CHANGED
|
@@ -4,12 +4,15 @@ import { readFile } from "node:fs/promises";
|
|
|
4
4
|
import * as path from "node:path";
|
|
5
5
|
import { pathToFileURL } from "node:url";
|
|
6
6
|
import { parseArgs } from "node:util";
|
|
7
|
+
import { isSeq, parse as parseYaml, parseDocument } from "yaml";
|
|
7
8
|
import { abortActiveRuns, ClaudeAdapter, DEFAULT_DIRECTIVE } from "./agents/claude.js";
|
|
8
9
|
import { BATCH_SCHEMA, buildBatchPrompt, parseBatchAnswer } from "./batch.js";
|
|
10
|
+
import { parseBaseline } from "./baseline.js";
|
|
9
11
|
import { Budget } from "./budget.js";
|
|
10
|
-
import { BUILTIN_SKILLS, ConfigError, findDefaultCasesFile, knownSkillNames, loadSuite, unknownNames, } from "./cases.js";
|
|
12
|
+
import { BUILTIN_SKILLS, ConfigError, findDefaultCasesFile, knownSkillNames, loadSuite, unknownNames, parseSuite, } from "./cases.js";
|
|
11
13
|
import { confusion } from "./confusion.js";
|
|
12
14
|
import { loadSkillDocs } from "./describe.js";
|
|
15
|
+
import { GEN_SCHEMA, buildGenPrompt, caseEntry, genSuite, orderCases, parseGenAnswer } from "./gen.js";
|
|
13
16
|
import { aggregate, judge } from "./judge.js";
|
|
14
17
|
import { evalSetToSuite } from "./import.js";
|
|
15
18
|
import { toJunit } from "./junit.js";
|
|
@@ -29,6 +32,7 @@ Usage:
|
|
|
29
32
|
skillcheck init [file] [options] write a starter cases file with those names
|
|
30
33
|
skillcheck import <file> --skill <name>
|
|
31
34
|
turn a skill-creator trigger eval set into cases
|
|
35
|
+
skillcheck gen [options] draft cases from skill descriptions (one model call per 8 skills)
|
|
32
36
|
|
|
33
37
|
run options:
|
|
34
38
|
-a, --agent <name> agent to route with (default: suite or claude)
|
|
@@ -47,6 +51,8 @@ run options:
|
|
|
47
51
|
and the terminal report to stderr)
|
|
48
52
|
--junit <path> JUnit XML report (GitLab/GitHub test reporters) to path
|
|
49
53
|
--markdown <path> Markdown summary (PR comments, GitHub job summary) to path
|
|
54
|
+
--baseline <path> earlier --json report: mark regressed, fixed and new cases
|
|
55
|
+
--only-new-failures exit 1 only for regressed or new failing cases, or budget skips (needs --baseline)
|
|
50
56
|
run/check options:
|
|
51
57
|
--skill <a,b> keep only cases that mention these skills
|
|
52
58
|
check options:
|
|
@@ -65,6 +71,15 @@ init options:
|
|
|
65
71
|
import options:
|
|
66
72
|
--skill <name> the skill the eval set is about
|
|
67
73
|
-o, --out <file> write the cases there instead of stdout (--force overwrites)
|
|
74
|
+
|
|
75
|
+
gen options:
|
|
76
|
+
--skill <a,b> skills to draft (default: user/project skills)
|
|
77
|
+
--plugin <name> skills of this installed plugin
|
|
78
|
+
--per-skill <n> positive requests per skill (default: 4)
|
|
79
|
+
-o, --out <file> write the cases there instead of stdout (--force overwrites)
|
|
80
|
+
--append <file> add the draft to an existing cases file (default: skills it lacks)
|
|
81
|
+
-m, -j, --timeout and --config-dir work as for run
|
|
82
|
+
|
|
68
83
|
common:
|
|
69
84
|
-h, --help show this help
|
|
70
85
|
--version show version
|
|
@@ -75,12 +90,12 @@ nothing reaches the model and nothing is billed, even when not logged in.
|
|
|
75
90
|
|
|
76
91
|
Exit codes: 0 all passed, 1 some case failed or was skipped, 2 config or environment error.
|
|
77
92
|
`;
|
|
78
|
-
const COMMANDS = ["run", "check", "lint", "list", "init", "import"];
|
|
93
|
+
const COMMANDS = ["run", "check", "lint", "list", "init", "import", "gen"];
|
|
79
94
|
const OPTIONS = {
|
|
80
95
|
help: { type: "boolean", short: "h", default: false },
|
|
81
96
|
version: { type: "boolean", default: false },
|
|
82
|
-
agent: { type: "string" },
|
|
83
|
-
model: { type: "string" },
|
|
97
|
+
agent: { type: "string", short: "a" },
|
|
98
|
+
model: { type: "string", short: "m" },
|
|
84
99
|
jobs: { type: "string", short: "j" },
|
|
85
100
|
only: { type: "string" },
|
|
86
101
|
repeat: { type: "string" },
|
|
@@ -96,22 +111,34 @@ const OPTIONS = {
|
|
|
96
111
|
json: { type: "string" },
|
|
97
112
|
junit: { type: "string" },
|
|
98
113
|
markdown: { type: "string" },
|
|
114
|
+
baseline: { type: "string" },
|
|
115
|
+
"only-new-failures": { type: "boolean", default: false },
|
|
99
116
|
top: { type: "string" },
|
|
100
117
|
overlap: { type: "string" },
|
|
101
118
|
strict: { type: "boolean", default: false },
|
|
102
119
|
"config-dir": { type: "string" },
|
|
103
120
|
force: { type: "boolean", default: false },
|
|
104
121
|
out: { type: "string", short: "o" },
|
|
122
|
+
"per-skill": { type: "string" },
|
|
123
|
+
plugin: { type: "string" },
|
|
124
|
+
append: { type: "string" },
|
|
105
125
|
};
|
|
106
126
|
/** Config/environment problems: message to stderr, exit 2. */
|
|
107
127
|
class UsageError extends Error {
|
|
108
128
|
}
|
|
109
129
|
function parseFlags(values, cwd) {
|
|
130
|
+
if (values.plugin !== undefined && values.skill !== undefined)
|
|
131
|
+
throw new UsageError("--plugin and --skill do not go together");
|
|
132
|
+
if (values.append !== undefined && values.out !== undefined)
|
|
133
|
+
throw new UsageError("--append and --out do not go together");
|
|
110
134
|
if (values["batch-size"] !== undefined && !values.batch)
|
|
111
135
|
throw new UsageError("--batch-size requires --batch");
|
|
112
136
|
if (values.batch && (values.directive !== undefined || values["no-early-stop"])) {
|
|
113
137
|
throw new UsageError("--directive and --no-early-stop do not apply to --batch");
|
|
114
138
|
}
|
|
139
|
+
if (values["only-new-failures"] && values.baseline === undefined) {
|
|
140
|
+
throw new UsageError("--only-new-failures requires --baseline");
|
|
141
|
+
}
|
|
115
142
|
return {
|
|
116
143
|
help: values.help,
|
|
117
144
|
version: values.version,
|
|
@@ -128,16 +155,21 @@ function parseFlags(values, cwd) {
|
|
|
128
155
|
batch: values.batch,
|
|
129
156
|
batchSize: intFlag(values["batch-size"], "batch-size", 1) ?? 25,
|
|
130
157
|
skill: values.skill,
|
|
158
|
+
plugin: values.plugin,
|
|
159
|
+
append: values.append,
|
|
131
160
|
budget: numberFlag(values.budget),
|
|
132
161
|
json: values.json,
|
|
133
162
|
junit: values.junit,
|
|
134
163
|
markdown: values.markdown,
|
|
164
|
+
baseline: values.baseline,
|
|
165
|
+
onlyNewFailures: values["only-new-failures"],
|
|
135
166
|
top: intFlag(values.top, "top", 1) ?? 5,
|
|
136
167
|
overlap: ratioFlag(values.overlap) ?? 0.3,
|
|
137
168
|
strict: values.strict,
|
|
138
169
|
configDir: values["config-dir"] !== undefined ? resolveConfigDir(values["config-dir"], cwd) : undefined,
|
|
139
170
|
force: values.force,
|
|
140
171
|
out: values.out,
|
|
172
|
+
perSkill: intFlag(values["per-skill"], "per-skill", 1) ?? 4,
|
|
141
173
|
};
|
|
142
174
|
}
|
|
143
175
|
/** --config-dir: absolute, must be an existing directory. */
|
|
@@ -225,6 +257,8 @@ export async function main(argv, io = { stdout: process.stdout, stderr: process.
|
|
|
225
257
|
return await initCommand(positionals[1], flags, io, deps);
|
|
226
258
|
if (command === "import")
|
|
227
259
|
return importCommand(positionals[1], flags, io);
|
|
260
|
+
if (command === "gen")
|
|
261
|
+
return await genCommand(flags, io, deps);
|
|
228
262
|
const file = casesFile(positionals[1], io);
|
|
229
263
|
if (command === "run")
|
|
230
264
|
return await runCommand(file, flags, io, deps);
|
|
@@ -428,6 +462,164 @@ function importCommand(positional, flags, io) {
|
|
|
428
462
|
io.stdout.write(`wrote ${flags.out} (${total} cases: ${suite.positive} should load ${skill}, ${suite.negative} should not)\n`);
|
|
429
463
|
return 0;
|
|
430
464
|
}
|
|
465
|
+
async function genCommand(flags, io, deps) {
|
|
466
|
+
const docs = loadSkillDocs({ cwd: io.cwd, configDir: flags.configDir });
|
|
467
|
+
const byName = new Map(docs.map((doc) => [doc.name, doc]));
|
|
468
|
+
const append = flags.append === undefined ? null : path.resolve(io.cwd, flags.append);
|
|
469
|
+
const existing = append === null ? null : await loadCases(append, io);
|
|
470
|
+
if (append !== null && existing === null)
|
|
471
|
+
return 2;
|
|
472
|
+
let targets;
|
|
473
|
+
if (flags.skill !== undefined) {
|
|
474
|
+
const names = flags.skill.split(",").map((name) => name.trim()).filter((name) => name !== "");
|
|
475
|
+
const unknown = names.filter((name, i) => !byName.has(name) && names.indexOf(name) === i);
|
|
476
|
+
if (unknown.length > 0)
|
|
477
|
+
throw new UsageError(`unknown skill: ${unknown.join(", ")}`);
|
|
478
|
+
targets = names.map((name) => byName.get(name)).filter((doc, i, all) => all.findIndex((other) => other.name === doc.name) === i);
|
|
479
|
+
}
|
|
480
|
+
else if (flags.plugin !== undefined) {
|
|
481
|
+
const pluginDocs = docs.filter((doc) => doc.plugin === flags.plugin);
|
|
482
|
+
if (pluginDocs.length === 0) {
|
|
483
|
+
const installed = [...new Set(docs.flatMap((doc) => doc.plugin === null ? [] : [doc.plugin]))].sort();
|
|
484
|
+
throw new UsageError(`unknown plugin: ${flags.plugin} (installed: ${installed.length > 0 ? installed.join(", ") : "none installed"})`);
|
|
485
|
+
}
|
|
486
|
+
targets = pluginDocs.filter((doc) => doc.kind === "skill");
|
|
487
|
+
if (targets.length === 0)
|
|
488
|
+
throw new UsageError(`plugin ${flags.plugin} has no skills`);
|
|
489
|
+
}
|
|
490
|
+
else if (append !== null && existing !== null) {
|
|
491
|
+
const covered = new Set(existing.cases.flatMap((item) => [...item.expect, ...item.expect_any, ...item.forbid, ...(item.first ? [item.first] : [])]));
|
|
492
|
+
targets = docs.filter((doc) => doc.kind === "skill" && doc.plugin === null && !covered.has(doc.name));
|
|
493
|
+
if (targets.length === 0) {
|
|
494
|
+
io.stderr.write(`nothing to draft: every skill already has cases in ${flags.append} (pass --skill to draft more)\n`);
|
|
495
|
+
return 0;
|
|
496
|
+
}
|
|
497
|
+
}
|
|
498
|
+
else {
|
|
499
|
+
targets = docs.filter((doc) => doc.kind === "skill" && doc.plugin === null);
|
|
500
|
+
}
|
|
501
|
+
if (targets.length === 0)
|
|
502
|
+
throw new UsageError("no skills found: put them in .claude/skills or pass --config-dir");
|
|
503
|
+
// before any model call: a refused file must not cost anything
|
|
504
|
+
const out = flags.out === undefined ? null : path.resolve(io.cwd, flags.out);
|
|
505
|
+
if (out !== null && fs.existsSync(out) && !flags.force)
|
|
506
|
+
throw new UsageError(`${flags.out} exists, use --force to overwrite`);
|
|
507
|
+
const adapter = pickAdapter(flags.agent ?? "claude", io, deps);
|
|
508
|
+
if (!adapter)
|
|
509
|
+
return 2;
|
|
510
|
+
if (!adapter.runBatch)
|
|
511
|
+
throw new UsageError(`agent ${adapter.name} cannot generate cases`);
|
|
512
|
+
const groups = [];
|
|
513
|
+
for (let i = 0; i < targets.length; i += 8)
|
|
514
|
+
groups.push(targets.slice(i, i + 8));
|
|
515
|
+
const results = await runPool(groups, flags.jobs, async (group) => {
|
|
516
|
+
try {
|
|
517
|
+
const result = await adapter.runBatch({
|
|
518
|
+
prompt: buildGenPrompt(group, docs, flags.perSkill),
|
|
519
|
+
schema: GEN_SCHEMA,
|
|
520
|
+
model: flags.model,
|
|
521
|
+
timeoutMs: flags.timeoutSec * 1000,
|
|
522
|
+
configDir: flags.configDir,
|
|
523
|
+
});
|
|
524
|
+
if (result.error) {
|
|
525
|
+
io.stderr.write(`skillcheck: gen failed for ${group.map((doc) => doc.name).join(", ")}: ${result.error}\n`);
|
|
526
|
+
return { cases: [], dropped: 0, costUsd: result.costUsd, failed: true };
|
|
527
|
+
}
|
|
528
|
+
const parsed = parseGenAnswer(result.structured, new Set(docs.map((doc) => doc.name)), new Set(group.map((doc) => doc.name)));
|
|
529
|
+
return { ...parsed, costUsd: result.costUsd, failed: false };
|
|
530
|
+
}
|
|
531
|
+
catch (e) {
|
|
532
|
+
const error = e.message;
|
|
533
|
+
io.stderr.write(`skillcheck: gen failed for ${group.map((doc) => doc.name).join(", ")}: ${error}\n`);
|
|
534
|
+
return { cases: [], dropped: 0, costUsd: null, failed: true };
|
|
535
|
+
}
|
|
536
|
+
});
|
|
537
|
+
const successful = results.filter((result) => result !== undefined && !result.failed);
|
|
538
|
+
if (successful.length === 0)
|
|
539
|
+
return 2;
|
|
540
|
+
// each answer is deduplicated on its own; groups can still repeat each other
|
|
541
|
+
const seen = new Set();
|
|
542
|
+
const cases = successful.flatMap((result) => result.cases).filter((item) => {
|
|
543
|
+
const key = item.query.trim();
|
|
544
|
+
if (seen.has(key))
|
|
545
|
+
return false;
|
|
546
|
+
seen.add(key);
|
|
547
|
+
return true;
|
|
548
|
+
});
|
|
549
|
+
const dropped = successful.reduce((total, result) => total + result.dropped, 0);
|
|
550
|
+
if (dropped > 0)
|
|
551
|
+
io.stderr.write(`note: dropped ${dropped} proposed cases naming unknown skills or none of the requested ones\n`);
|
|
552
|
+
const cost = genCost(results.map((result) => result?.costUsd ?? null));
|
|
553
|
+
if (cases.length === 0) {
|
|
554
|
+
io.stderr.write(`skillcheck: gen got no usable cases from the model (cost $${cost})\n`);
|
|
555
|
+
return 2;
|
|
556
|
+
}
|
|
557
|
+
if (append !== null) {
|
|
558
|
+
const oldQueries = new Set(existing.cases.map((item) => item.query.trim()));
|
|
559
|
+
const unique = cases.filter((item) => !oldQueries.has(item.query.trim()));
|
|
560
|
+
const duplicates = cases.length - unique.length;
|
|
561
|
+
if (unique.length > 0)
|
|
562
|
+
appendCases(append, unique, targets.map((doc) => doc.name), flags.model ?? null);
|
|
563
|
+
const suffix = duplicates > 0 ? `, ${duplicates} duplicates skipped` : "";
|
|
564
|
+
io.stdout.write(`appended ${unique.length} cases to ${flags.append} (${targets.length} skills${suffix}, cost $${cost})\n`);
|
|
565
|
+
return 0;
|
|
566
|
+
}
|
|
567
|
+
const text = genSuite(cases, { model: flags.model ?? null, skills: targets.map((doc) => doc.name), json: out !== null && path.extname(out).toLowerCase() === ".json" });
|
|
568
|
+
const counts = `${cases.length} cases for ${targets.length} skills, cost $${cost}`;
|
|
569
|
+
if (out === null) {
|
|
570
|
+
io.stdout.write(text);
|
|
571
|
+
io.stderr.write(`drafted ${counts}\n`);
|
|
572
|
+
return 0;
|
|
573
|
+
}
|
|
574
|
+
try {
|
|
575
|
+
fs.writeFileSync(out, text);
|
|
576
|
+
}
|
|
577
|
+
catch (e) {
|
|
578
|
+
throw new UsageError(`cannot write ${flags.out}: ${e.message}`);
|
|
579
|
+
}
|
|
580
|
+
io.stdout.write(`wrote ${flags.out} (${counts})\n`);
|
|
581
|
+
return 0;
|
|
582
|
+
}
|
|
583
|
+
function genCost(costs) {
|
|
584
|
+
const known = costs.filter((cost) => cost !== null);
|
|
585
|
+
const unknown = costs.length - known.length;
|
|
586
|
+
return known.length === 0 ? "?" : known.reduce((total, value) => total + value, 0).toFixed(2) +
|
|
587
|
+
(unknown > 0 ? ` + ${unknown} call${unknown === 1 ? "" : "s"} of unknown cost` : "");
|
|
588
|
+
}
|
|
589
|
+
function appendCases(file, cases, skills, model) {
|
|
590
|
+
const ext = path.extname(file).toLowerCase();
|
|
591
|
+
const ordered = orderCases(cases, skills);
|
|
592
|
+
const text = fs.readFileSync(file, "utf8");
|
|
593
|
+
let nextText;
|
|
594
|
+
if (ext === ".json") {
|
|
595
|
+
const data = JSON.parse(text);
|
|
596
|
+
if (Array.isArray(data))
|
|
597
|
+
data.push(...ordered.map(caseEntry));
|
|
598
|
+
else
|
|
599
|
+
data.cases.push(...ordered.map(caseEntry));
|
|
600
|
+
nextText = JSON.stringify(data, null, 2) + "\n";
|
|
601
|
+
}
|
|
602
|
+
else {
|
|
603
|
+
const doc = parseDocument(text);
|
|
604
|
+
// a bare list of cases is the older format, still accepted
|
|
605
|
+
const sequence = (isSeq(doc.contents) ? doc.contents : doc.get("cases", true));
|
|
606
|
+
for (const [i, item] of ordered.entries()) {
|
|
607
|
+
const node = doc.createNode(caseEntry(item));
|
|
608
|
+
// [name] like the rest of a cases file, not a block list
|
|
609
|
+
for (const key of ["expect", "forbid"]) {
|
|
610
|
+
const list = node.get(key, true);
|
|
611
|
+
if (isSeq(list))
|
|
612
|
+
list.flow = true;
|
|
613
|
+
}
|
|
614
|
+
if (i === 0)
|
|
615
|
+
node.commentBefore = ` gen draft (model: ${model ?? "default"}): review these cases`;
|
|
616
|
+
sequence.add(node);
|
|
617
|
+
}
|
|
618
|
+
nextText = doc.toString({ flowCollectionPadding: false });
|
|
619
|
+
}
|
|
620
|
+
parseSuite(ext === ".json" ? JSON.parse(nextText) : parseYaml(nextText));
|
|
621
|
+
fs.writeFileSync(file, nextText);
|
|
622
|
+
}
|
|
431
623
|
const UNCOVERED_WIDTH = 78;
|
|
432
624
|
function printLint(out, report, docs, suite) {
|
|
433
625
|
const casesPart = suite ? `${report.cases} cases` : "no cases file";
|
|
@@ -503,6 +695,15 @@ async function runCommand(file, flags, io, deps) {
|
|
|
503
695
|
const suite = await loadCases(file, io);
|
|
504
696
|
if (!suite)
|
|
505
697
|
return 2;
|
|
698
|
+
let baseline;
|
|
699
|
+
if (flags.baseline !== undefined) {
|
|
700
|
+
try {
|
|
701
|
+
baseline = parseBaseline(fs.readFileSync(path.resolve(io.cwd, flags.baseline), "utf8"));
|
|
702
|
+
}
|
|
703
|
+
catch (e) {
|
|
704
|
+
throw new UsageError(`cannot read baseline ${flags.baseline}: ${e.message}`);
|
|
705
|
+
}
|
|
706
|
+
}
|
|
506
707
|
const agent = flags.agent ?? suite.agent;
|
|
507
708
|
const adapter = pickAdapter(agent, io, deps);
|
|
508
709
|
if (!adapter)
|
|
@@ -546,6 +747,24 @@ async function runCommand(file, flags, io, deps) {
|
|
|
546
747
|
const expected = expectedNames(selected);
|
|
547
748
|
const unavailable = outcome.sawAvailability ? expected.filter((name) => !outcome.available.has(name)) : [];
|
|
548
749
|
const pairs = confusion(ordered);
|
|
750
|
+
const report = buildReport({
|
|
751
|
+
file,
|
|
752
|
+
agent,
|
|
753
|
+
model: model ?? null,
|
|
754
|
+
startedAtMs,
|
|
755
|
+
durationMs: Date.now() - startedAtMs,
|
|
756
|
+
cases: items.map((item) => ({ c: item.c, threshold: item.threshold, result: outcome.results[item.jobIndex] ?? null })),
|
|
757
|
+
unavailable,
|
|
758
|
+
confusion: pairs,
|
|
759
|
+
batch: flags.batch,
|
|
760
|
+
estimatedCostUsd: budget.spent,
|
|
761
|
+
skippedRunCosts,
|
|
762
|
+
budgetUsd: flags.budget ?? null,
|
|
763
|
+
budgetReached: skipped.length > 0,
|
|
764
|
+
baseline,
|
|
765
|
+
baselineFile: flags.baseline,
|
|
766
|
+
suiteCases: suite.cases.map((c) => ({ id: c.id ?? null, query: c.query })),
|
|
767
|
+
});
|
|
549
768
|
reporter.summary(ordered, {
|
|
550
769
|
unavailable,
|
|
551
770
|
confusion: pairs,
|
|
@@ -553,6 +772,11 @@ async function runCommand(file, flags, io, deps) {
|
|
|
553
772
|
budget: skipped.length > 0 ? { limitUsd: flags.budget, spent: budget.spent, notStartedRuns } : undefined,
|
|
554
773
|
estimatedUsd: budget.spent,
|
|
555
774
|
skippedRunCosts,
|
|
775
|
+
baseline: report.baseline ? {
|
|
776
|
+
summary: report.baseline,
|
|
777
|
+
regressed: report.cases.filter((c) => c.change === "regressed").map((c) => `#${c.id ?? c.index}`),
|
|
778
|
+
fixed: report.cases.filter((c) => c.change === "fixed").map((c) => `#${c.id ?? c.index}`),
|
|
779
|
+
} : undefined,
|
|
556
780
|
});
|
|
557
781
|
if (flags.batch) {
|
|
558
782
|
// cases that only errored have no answer to confirm
|
|
@@ -562,21 +786,12 @@ async function runCommand(file, flags, io, deps) {
|
|
|
562
786
|
const hint = failed.length > 0 ? ` (--only ${failed.join(",")})` : "";
|
|
563
787
|
reporter.note(`batch mode: answers are the model's stated choice, not an actual Skill call — confirm failures with a normal run${hint}`);
|
|
564
788
|
}
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
cases: items.map((item) => ({ c: item.c, threshold: item.threshold, result: outcome.results[item.jobIndex] ?? null })),
|
|
572
|
-
unavailable,
|
|
573
|
-
confusion: pairs,
|
|
574
|
-
batch: flags.batch,
|
|
575
|
-
estimatedCostUsd: budget.spent,
|
|
576
|
-
skippedRunCosts,
|
|
577
|
-
budgetUsd: flags.budget ?? null,
|
|
578
|
-
budgetReached: skipped.length > 0,
|
|
579
|
-
});
|
|
789
|
+
if (baseline && baseline.model !== null && report.model !== null && baseline.model !== report.model) {
|
|
790
|
+
reporter.note(`baseline was run with model ${baseline.model}, this run with ${report.model}`);
|
|
791
|
+
}
|
|
792
|
+
if (baseline && baseline.batch !== report.batch) {
|
|
793
|
+
reporter.note(baseline.batch ? "baseline was a batch run, this one is not" : "this is a batch run, the baseline was not");
|
|
794
|
+
}
|
|
580
795
|
if (flags.json !== undefined) {
|
|
581
796
|
const payload = JSON.stringify(report, null, 2) + "\n";
|
|
582
797
|
if (flags.json === "-")
|
|
@@ -588,6 +803,9 @@ async function runCommand(file, flags, io, deps) {
|
|
|
588
803
|
writeReportFile(flags.junit, toJunit(report), "--junit");
|
|
589
804
|
if (flags.markdown !== undefined)
|
|
590
805
|
writeReportFile(flags.markdown, toMarkdown(report), "--markdown");
|
|
806
|
+
if (flags.onlyNewFailures) {
|
|
807
|
+
return report.cases.some((c) => c.status === "skipped" || (c.status === "failed" && (c.change === "regressed" || c.change === "new"))) ? 1 : 0;
|
|
808
|
+
}
|
|
591
809
|
return ordered.some((r) => !r.ok) || skipped.length > 0 ? 1 : 0;
|
|
592
810
|
}
|
|
593
811
|
function writeReportFile(path, text, flag) {
|
package/dist/gen.js
ADDED
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
import { parse as parseYaml } from "yaml";
|
|
2
|
+
import { parseSuite } from "./cases.js";
|
|
3
|
+
import { yamlName } from "./import.js";
|
|
4
|
+
export const GEN_SCHEMA = {
|
|
5
|
+
type: "object",
|
|
6
|
+
properties: {
|
|
7
|
+
cases: {
|
|
8
|
+
type: "array",
|
|
9
|
+
items: {
|
|
10
|
+
type: "object",
|
|
11
|
+
properties: {
|
|
12
|
+
query: { type: "string" },
|
|
13
|
+
skill: { type: ["string", "null"] },
|
|
14
|
+
avoid: { type: ["string", "null"] },
|
|
15
|
+
},
|
|
16
|
+
required: ["query", "skill", "avoid"],
|
|
17
|
+
},
|
|
18
|
+
},
|
|
19
|
+
},
|
|
20
|
+
required: ["cases"],
|
|
21
|
+
};
|
|
22
|
+
/** Order generated cases by their requested skills and convert one to a suite entry. */
|
|
23
|
+
export function orderCases(cases, skills) {
|
|
24
|
+
const buckets = new Map(skills.map((skill) => [skill, []]));
|
|
25
|
+
const rest = [];
|
|
26
|
+
for (const item of cases) {
|
|
27
|
+
const target = item.avoid ?? item.skill;
|
|
28
|
+
const bucket = target === null ? undefined : buckets.get(target);
|
|
29
|
+
if (bucket)
|
|
30
|
+
bucket.push(item);
|
|
31
|
+
else
|
|
32
|
+
rest.push(item);
|
|
33
|
+
}
|
|
34
|
+
return [...skills.flatMap((skill) => buckets.get(skill) ?? []), ...rest];
|
|
35
|
+
}
|
|
36
|
+
/** A near miss owned by a neighbour gets both: expect the neighbour, forbid the target. */
|
|
37
|
+
export function caseEntry(item) {
|
|
38
|
+
return {
|
|
39
|
+
query: item.query,
|
|
40
|
+
...(item.skill !== null ? { expect: [item.skill] } : {}),
|
|
41
|
+
...(item.avoid !== null ? { forbid: [item.avoid] } : {}),
|
|
42
|
+
};
|
|
43
|
+
}
|
|
44
|
+
/** Build the case-writing prompt from installed descriptions. */
|
|
45
|
+
export function buildGenPrompt(targets, context, perSkill) {
|
|
46
|
+
const contextLines = context.map((doc) => `- ${doc.name}: ${JSON.stringify(doc.description.replace(/\s+/g, " ").slice(0, 300))}`);
|
|
47
|
+
const near = Math.ceil(perSkill / 2);
|
|
48
|
+
const targetLines = targets.map((doc) => `For ${doc.name}: write ${perSkill} positive requests and ${near} near misses. ` +
|
|
49
|
+
`Positive requests have skill ${JSON.stringify(doc.name)} and avoid null. ` +
|
|
50
|
+
`Near misses have avoid ${JSON.stringify(doc.name)}, and skill is another listed skill or null.`);
|
|
51
|
+
return [
|
|
52
|
+
"This is a test-writing task. Do not load any skill, do not use tools other than the structured output.",
|
|
53
|
+
"Context skills:",
|
|
54
|
+
...contextLines,
|
|
55
|
+
"For each target skill, write realistic first messages a user would send. They should vary in wording and length, must not name the skill, and should be written in the language of the skill description.",
|
|
56
|
+
...targetLines,
|
|
57
|
+
"Return only the structured output. Each request must be a JSON string.",
|
|
58
|
+
].join("\n");
|
|
59
|
+
}
|
|
60
|
+
// targets: the skills asked for; every case must load or avoid one of them
|
|
61
|
+
export function parseGenAnswer(structured, known, targets = known) {
|
|
62
|
+
if (typeof structured !== "object" || structured === null || Array.isArray(structured))
|
|
63
|
+
return { cases: [], dropped: 0 };
|
|
64
|
+
const rawCases = structured.cases;
|
|
65
|
+
if (!Array.isArray(rawCases))
|
|
66
|
+
return { cases: [], dropped: 0 };
|
|
67
|
+
const cases = [];
|
|
68
|
+
const seen = new Set();
|
|
69
|
+
let dropped = 0;
|
|
70
|
+
for (const raw of rawCases) {
|
|
71
|
+
if (typeof raw !== "object" || raw === null || Array.isArray(raw)) {
|
|
72
|
+
dropped++;
|
|
73
|
+
continue;
|
|
74
|
+
}
|
|
75
|
+
const item = raw;
|
|
76
|
+
const query = item.query;
|
|
77
|
+
const skill = item.skill === null ? null : item.skill;
|
|
78
|
+
const avoid = item.avoid === null ? null : item.avoid;
|
|
79
|
+
if (typeof query !== "string" || query.trim() === "" ||
|
|
80
|
+
(skill !== null && (typeof skill !== "string" || !known.has(skill))) ||
|
|
81
|
+
(avoid !== null && (typeof avoid !== "string" || !known.has(avoid))) ||
|
|
82
|
+
(skill === null && avoid === null) || skill === avoid || seen.has(query) ||
|
|
83
|
+
!((skill !== null && targets.has(skill)) || (avoid !== null && targets.has(avoid)))) {
|
|
84
|
+
dropped++;
|
|
85
|
+
continue;
|
|
86
|
+
}
|
|
87
|
+
seen.add(query);
|
|
88
|
+
cases.push({ query, skill: skill, avoid: avoid });
|
|
89
|
+
}
|
|
90
|
+
return { cases, dropped };
|
|
91
|
+
}
|
|
92
|
+
export function genSuite(cases, meta) {
|
|
93
|
+
const ordered = orderCases(cases, meta.skills);
|
|
94
|
+
if (meta.json) {
|
|
95
|
+
// JSON has no comments: the review note lives only in the command output
|
|
96
|
+
const data = { agent: "claude", repeat: 1, threshold: 1, cases: ordered.map(caseEntry) };
|
|
97
|
+
parseSuite(data);
|
|
98
|
+
return JSON.stringify(data, null, 2) + "\n";
|
|
99
|
+
}
|
|
100
|
+
const lines = [
|
|
101
|
+
`# Draft cases written by skillcheck gen (model: ${meta.model ?? "default"}) for: ${meta.skills.join(", ")}.`,
|
|
102
|
+
"# Review every case: the model guessed what should route where. Delete what is wrong,",
|
|
103
|
+
"# keep what matches how people really ask, then `skillcheck run --batch`.",
|
|
104
|
+
"agent: claude",
|
|
105
|
+
"repeat: 1",
|
|
106
|
+
"threshold: 1.0",
|
|
107
|
+
"cases:",
|
|
108
|
+
];
|
|
109
|
+
for (const item of ordered) {
|
|
110
|
+
lines.push(` - query: ${JSON.stringify(item.query)}`);
|
|
111
|
+
if (item.skill !== null)
|
|
112
|
+
lines.push(` expect: [${yamlName(item.skill)}]`);
|
|
113
|
+
if (item.avoid !== null)
|
|
114
|
+
lines.push(` forbid: [${yamlName(item.avoid)}]`);
|
|
115
|
+
}
|
|
116
|
+
const text = lines.join("\n") + "\n";
|
|
117
|
+
parseSuite(parseYaml(text));
|
|
118
|
+
return text;
|
|
119
|
+
}
|
package/dist/import.js
CHANGED
|
@@ -2,6 +2,9 @@ import * as path from "node:path";
|
|
|
2
2
|
import { parse as parseYaml } from "yaml";
|
|
3
3
|
import { ConfigError, parseSuite } from "./cases.js";
|
|
4
4
|
const PLAIN_NAME = /^[A-Za-z0-9._:/-]+$/;
|
|
5
|
+
export function yamlName(name) {
|
|
6
|
+
return PLAIN_NAME.test(name) && parseYaml(name) === name ? name : JSON.stringify(name);
|
|
7
|
+
}
|
|
5
8
|
/**
|
|
6
9
|
* A skill-creator trigger eval set ([{query, should_trigger}], one skill) as a
|
|
7
10
|
* cases file: should_trigger true becomes expect, false becomes forbid.
|
|
@@ -29,7 +32,7 @@ export function evalSetToSuite(data, skill, source) {
|
|
|
29
32
|
if (errors.length > 0)
|
|
30
33
|
throw new ConfigError(errors);
|
|
31
34
|
// plain only when YAML reads it back as the same string: not true, null, 123
|
|
32
|
-
const name =
|
|
35
|
+
const name = yamlName(skill);
|
|
33
36
|
const lines = [
|
|
34
37
|
`# skillcheck cases imported from ${path.basename(source)} (skill-creator trigger eval set).`,
|
|
35
38
|
"# should_trigger: true became expect, false became forbid: another skill may",
|
package/dist/junit.js
CHANGED
|
@@ -19,6 +19,12 @@ export function toJunit(report) {
|
|
|
19
19
|
properties.push(` <property name="unknownCostRuns" value="${report.summary.unknownCostRuns}"/>`);
|
|
20
20
|
properties.push(` <property name="estimatedCostUsd" value="${report.summary.estimatedCostUsd.toFixed(2)}"/>`);
|
|
21
21
|
}
|
|
22
|
+
if (report.baseline) {
|
|
23
|
+
properties.push(` <property name="baselineRegressed" value="${report.baseline.regressed}"/>`);
|
|
24
|
+
properties.push(` <property name="baselineFixed" value="${report.baseline.fixed}"/>`);
|
|
25
|
+
properties.push(` <property name="baselineNew" value="${report.baseline.new}"/>`);
|
|
26
|
+
properties.push(` <property name="baselineRemoved" value="${report.baseline.removed}"/>`);
|
|
27
|
+
}
|
|
22
28
|
properties.push(" </properties>");
|
|
23
29
|
const cases = rows.map(({ c, kind }) => testcase(c, kind, suiteName));
|
|
24
30
|
return [
|
package/dist/markdown.js
CHANGED
|
@@ -6,6 +6,10 @@ export function toMarkdown(report) {
|
|
|
6
6
|
const lines = [MARKDOWN_MARKER, header(report)];
|
|
7
7
|
lines.push("");
|
|
8
8
|
lines.push(metaLine(report));
|
|
9
|
+
if (report.baseline) {
|
|
10
|
+
lines.push("");
|
|
11
|
+
lines.push(`> Since baseline: ${baselineText(report.baseline)}`);
|
|
12
|
+
}
|
|
9
13
|
if (report.batch) {
|
|
10
14
|
lines.push("");
|
|
11
15
|
lines.push("> Batch mode: answers are the model's stated choice, not an actual Skill call. Confirm failures with a normal run.");
|
|
@@ -101,6 +105,10 @@ function reasonCell(c) {
|
|
|
101
105
|
const bad = c.runs.filter((r) => !r.ok);
|
|
102
106
|
const parts = c.runs.length > 1 ? [`${c.passed}/${c.runs.length}`, bad[0]?.reason ?? ""] : [bad[0]?.reason ?? ""];
|
|
103
107
|
let text = parts.join(" · ");
|
|
108
|
+
if (c.change === "regressed")
|
|
109
|
+
text = `**regressed** · ${text}`;
|
|
110
|
+
else if (c.change === "new")
|
|
111
|
+
text = `new · ${text}`;
|
|
104
112
|
const diagnosis = bad.find((r) => r.diagnosis !== null)?.diagnosis;
|
|
105
113
|
if (diagnosis)
|
|
106
114
|
text += ` — _${diagnosis}_`;
|
|
@@ -110,9 +118,18 @@ function reasonCell(c) {
|
|
|
110
118
|
}
|
|
111
119
|
function passedRow(c) {
|
|
112
120
|
const score = c.runs.length > 1 ? ` (${c.passed}/${c.runs.length})` : "";
|
|
113
|
-
const cells = [`${caseLabel(c)}${score}`, oneLine(c.query), loaded(c)];
|
|
121
|
+
const cells = [`${caseLabel(c)}${score}`, oneLine(c.query), `${loaded(c)}${c.change === "fixed" ? " · fixed" : ""}`];
|
|
114
122
|
return `| ${cells.map(esc).join(" | ")} |`;
|
|
115
123
|
}
|
|
124
|
+
function baselineText(summary) {
|
|
125
|
+
const parts = [
|
|
126
|
+
summary.regressed > 0 ? `**${summary.regressed} regressed**` : null,
|
|
127
|
+
summary.fixed > 0 ? `${summary.fixed} fixed` : null,
|
|
128
|
+
summary.new > 0 ? `${summary.new} new` : null,
|
|
129
|
+
summary.removed > 0 ? `${summary.removed} removed` : null,
|
|
130
|
+
].filter((p) => p !== null);
|
|
131
|
+
return parts.length > 0 ? `${parts.join(", ")}.` : "no changes.";
|
|
132
|
+
}
|
|
116
133
|
function caseLabel(c) {
|
|
117
134
|
return `#${c.id ?? c.index}`;
|
|
118
135
|
}
|
package/dist/report.js
CHANGED
|
@@ -72,6 +72,16 @@ export class Reporter {
|
|
|
72
72
|
const skippedPart = skipped > 0 ? `, ${skipped} skipped` : "";
|
|
73
73
|
const costs = [...verdicts.map((v) => v.costUsd), ...(extra?.skippedRunCosts ?? [])];
|
|
74
74
|
this.write(`${failed} failed${skippedPart} of ${results.length + skipped} · runs ${costs.length} · ${costLine(costs, extra?.estimatedUsd)}\n`);
|
|
75
|
+
if (extra?.baseline) {
|
|
76
|
+
const b = extra.baseline;
|
|
77
|
+
const parts = [
|
|
78
|
+
b.summary.regressed > 0 ? this.paint("31", `${b.summary.regressed} regressed (${b.regressed.join(", ")})`) : null,
|
|
79
|
+
b.summary.fixed > 0 ? this.paint("32", `${b.summary.fixed} fixed (${b.fixed.join(", ")})`) : null,
|
|
80
|
+
b.summary.new > 0 ? `${b.summary.new} new` : null,
|
|
81
|
+
b.summary.removed > 0 ? `${b.summary.removed} removed` : null,
|
|
82
|
+
].filter((p) => p !== null);
|
|
83
|
+
this.write(`vs baseline: ${parts.length > 0 ? parts.join(", ") : "no changes"}\n`);
|
|
84
|
+
}
|
|
75
85
|
}
|
|
76
86
|
paint(code, text) {
|
|
77
87
|
return this.useColor ? `\u001b[${code}m${text}${RESET}` : text;
|
package/dist/results.js
CHANGED
|
@@ -1,9 +1,12 @@
|
|
|
1
|
+
import { compare } from "./baseline.js";
|
|
1
2
|
import { readVersion } from "./version.js";
|
|
2
3
|
/** The single source of truth for --json and --junit; computed once per run. */
|
|
3
4
|
export function buildReport(input) {
|
|
4
5
|
const verdicts = input.cases.flatMap((e) => e.result?.runs ?? []);
|
|
5
6
|
const costs = [...verdicts.map((v) => v.costUsd), ...(input.skippedRunCosts ?? [])];
|
|
6
7
|
const known = costs.filter((c) => c !== null);
|
|
8
|
+
const cases = input.cases.map(toCaseReport);
|
|
9
|
+
const comparison = input.baseline ? compare(cases, input.baseline, input.baselineFile ?? input.baseline.file, input.suiteCases ?? cases) : null;
|
|
7
10
|
return {
|
|
8
11
|
tool: "skillcheck",
|
|
9
12
|
version: readVersion(),
|
|
@@ -27,7 +30,8 @@ export function buildReport(input) {
|
|
|
27
30
|
},
|
|
28
31
|
unavailable: input.unavailable,
|
|
29
32
|
confusion: input.confusion,
|
|
30
|
-
cases:
|
|
33
|
+
cases: comparison ? cases.map((c, i) => ({ ...c, change: comparison.changes[i] ?? null })) : cases,
|
|
34
|
+
baseline: comparison?.summary ?? null,
|
|
31
35
|
};
|
|
32
36
|
}
|
|
33
37
|
function toCaseReport(e) {
|
|
@@ -46,5 +50,6 @@ function toCaseReport(e) {
|
|
|
46
50
|
passed: e.result?.passed ?? 0,
|
|
47
51
|
threshold: e.result?.threshold ?? e.threshold,
|
|
48
52
|
runs: e.result?.runs ?? [],
|
|
53
|
+
change: null,
|
|
49
54
|
};
|
|
50
55
|
}
|