@icntswm/skillcheck 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +149 -0
- package/dist/agents/claude.js +384 -0
- package/dist/agents/types.js +1 -0
- package/dist/batch.js +111 -0
- package/dist/budget.js +29 -0
- package/dist/cases.js +249 -0
- package/dist/cli.js +673 -0
- package/dist/confusion.js +45 -0
- package/dist/describe.js +282 -0
- package/dist/judge.js +53 -0
- package/dist/junit.js +84 -0
- package/dist/lexical.js +117 -0
- package/dist/lint.js +51 -0
- package/dist/pool.js +35 -0
- package/dist/report.js +93 -0
- package/dist/results.js +48 -0
- package/dist/version.js +7 -0
- package/package.json +47 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 icntswm
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
ADDED
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
# skillcheck
|
|
2
|
+
|
|
3
|
+
[](LICENSE)
|
|
4
|
+
[](https://www.npmjs.com/package/@icntswm/skillcheck)
|
|
5
|
+

|
|
6
|
+

|
|
7
|
+
|
|
8
|
+
**Regression tests for Claude Code skills.**
|
|
9
|
+
|
|
10
|
+
You edited a skill description, installed a plugin or switched models, and now
|
|
11
|
+
some of your requests load the wrong skill. Nothing warns you: the agent still
|
|
12
|
+
answers, just with the wrong instructions. skillcheck catches this before your
|
|
13
|
+
users do.
|
|
14
|
+
|
|
15
|
+
```
|
|
16
|
+
$ skillcheck run examples/demo/skillcheck.yaml --skill test-guard,find-bug
|
|
17
|
+
5 cases × 1 repeat × 1 agent = 5 runs
|
|
18
|
+
ok #bug-500 POST /orders returns 500 with 'nil pointer deref… → find-bug
|
|
19
|
+
FAIL #flaky-ci-only this test is red in CI roughly once a day but I … → find-bug · not loaded test-guard
|
|
20
|
+
FAIL #flaky-retry TestCartMerge fails sometimes and goes green w… → find-bug · not loaded test-guard; forbidden find-bug
|
|
21
|
+
ok #bug-consistent TestCheckoutTotal fails on every run since this … → find-bug
|
|
22
|
+
ok #perf-latency the /search endpoint went from 80ms to 900ms aft… → perf-profile
|
|
23
|
+
|
|
24
|
+
confusion:
|
|
25
|
+
expected test-guard → got find-bug (2)
|
|
26
|
+
2 failed of 5 · runs 5 · cost $0.25
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
This is a real run from the [demo](examples/demo): one skill's description got
|
|
30
|
+
wider, and it started taking requests that belong to its neighbour.
|
|
31
|
+
|
|
32
|
+
## What you get
|
|
33
|
+
|
|
34
|
+
- 🎯 **The real routing decision.** skillcheck runs Claude Code itself and reads
|
|
35
|
+
which skills the model actually loads. Nothing is simulated or guessed from
|
|
36
|
+
keywords.
|
|
37
|
+
- 💸 **Cheap by design.** Free static checks first. Then one batch call covers
|
|
38
|
+
25 requests. A full run stops Claude Code the moment it picks a skill, so
|
|
39
|
+
you pay for the decision, not for the work.
|
|
40
|
+
- 🔍 **Points at the fix.** The confusion block shows which skill took whose
|
|
41
|
+
requests. When the model names the right skill but doesn't load it, the
|
|
42
|
+
report says so separately.
|
|
43
|
+
- 🛡️ **Safe on any project.** Runs go in plan mode with edits, shell, web and
|
|
44
|
+
MCP tools disabled. Your files are never touched.
|
|
45
|
+
- 🧪 **Honest about randomness.** `--repeat` and `threshold` tell a flaky case
|
|
46
|
+
from a broken one instead of letting you guess.
|
|
47
|
+
- 🏷️ **Catches renames and typos.** Case names are checked against the skills
|
|
48
|
+
the agent really has, so a renamed skill can't pass silently.
|
|
49
|
+
- ⚙️ **CI-ready.** JUnit and JSON reports, clear exit codes, a spending cap
|
|
50
|
+
(`--budget`), and `--config-dir` to test only the skills in your repository.
|
|
51
|
+
- 📄 **Plain YAML, one dependency.** Cases are readable by anyone on the team
|
|
52
|
+
and live next to the skills they test.
|
|
53
|
+
|
|
54
|
+
## When to run it
|
|
55
|
+
|
|
56
|
+
- after you rewrite or shorten a skill description;
|
|
57
|
+
- when you add a skill that sounds like an existing one;
|
|
58
|
+
- after installing a plugin that brings its own skills;
|
|
59
|
+
- before switching to another model;
|
|
60
|
+
- on every pull request that touches `skills/`.
|
|
61
|
+
|
|
62
|
+
## How it works in one minute
|
|
63
|
+
|
|
64
|
+
Write requests the way you actually type them, and say which skill must, or
|
|
65
|
+
must not, load:
|
|
66
|
+
|
|
67
|
+
```yaml
|
|
68
|
+
cases:
|
|
69
|
+
- query: "why does TestOrderCreate fail? it was green yesterday"
|
|
70
|
+
expect: [find-bug]
|
|
71
|
+
forbid: [test-guard]
|
|
72
|
+
- query: "what is 2 + 2?"
|
|
73
|
+
none: true
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
skillcheck sends each request to a headless `claude -p`, watches which skills
|
|
77
|
+
the model loads, and stops the process as soon as the answer is clear.
|
|
78
|
+
|
|
79
|
+
## Three levels, from free to exact
|
|
80
|
+
|
|
81
|
+
Start at the cheapest level and go up only when it finds nothing.
|
|
82
|
+
|
|
83
|
+
| Command | Model calls | What it tells you |
|
|
84
|
+
|---|---|---|
|
|
85
|
+
| `skillcheck lint` | none | short descriptions, look-alike skills, cases whose expected skill shares few words with the request |
|
|
86
|
+
| `skillcheck run --batch` | one per 25 cases | which skill the model *says* it would load |
|
|
87
|
+
| `skillcheck run` | one per case | which skill the model *actually* loads |
|
|
88
|
+
|
|
89
|
+
On the demo suite one batch call covered all 12 cases for $0.05 and agreed
|
|
90
|
+
with the normal run in 34 checks out of 34. More in [docs/cost.md](docs/cost.md).
|
|
91
|
+
|
|
92
|
+
## Does it really catch regressions?
|
|
93
|
+
|
|
94
|
+
Yes, and the [demo](examples/demo) shows it: six skills, twelve cases and a bad
|
|
95
|
+
edit of two descriptions.
|
|
96
|
+
|
|
97
|
+
| | good skills | bad edit |
|
|
98
|
+
|---|---|---|
|
|
99
|
+
| `lint` | no problems | flags the too-short description |
|
|
100
|
+
| `run --batch` | 12/12 passed | catches both misrouted cases |
|
|
101
|
+
| `run` | 5/5 passed | catches the same two cases |
|
|
102
|
+
|
|
103
|
+
The demo README also lists edits that did *not* break routing, and why.
|
|
104
|
+
|
|
105
|
+
## Install
|
|
106
|
+
|
|
107
|
+
Needs Node 20+ and [Claude Code](https://docs.anthropic.com/en/docs/claude-code)
|
|
108
|
+
(`claude`) in `PATH`, logged in.
|
|
109
|
+
|
|
110
|
+
```
|
|
111
|
+
npm install -g @icntswm/skillcheck
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
The command is `skillcheck`. To try it without installing:
|
|
115
|
+
`npx @icntswm/skillcheck lint`.
|
|
116
|
+
|
|
117
|
+
## Quick start
|
|
118
|
+
|
|
119
|
+
```
|
|
120
|
+
skillcheck init # writes skillcheck.yaml listing your skills
|
|
121
|
+
# then add a few real requests per skill
|
|
122
|
+
skillcheck check # validates the file, no model calls
|
|
123
|
+
skillcheck lint # free static checks
|
|
124
|
+
skillcheck run --batch # cheap pre-check, one call
|
|
125
|
+
skillcheck run --only 2,5 # confirm what the batch flagged
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
`skillcheck list` prints the skills and slash commands Claude Code sees. It
|
|
129
|
+
stops Claude Code before the first model call, so it costs nothing.
|
|
130
|
+
|
|
131
|
+
## Documentation
|
|
132
|
+
|
|
133
|
+
| | |
|
|
134
|
+
|---|---|
|
|
135
|
+
| [Writing cases](docs/writing-cases.md) | the file format and what makes a case catch regressions |
|
|
136
|
+
| [Cost](docs/cost.md) | what a run costs and how to spend less |
|
|
137
|
+
| [Reports and CI](docs/ci.md) | confusion block, JSON, JUnit, GitHub Actions, exit codes |
|
|
138
|
+
| [How it works](docs/how-it-works.md) | what happens inside a run, and the limits of each level |
|
|
139
|
+
|
|
140
|
+
`skillcheck --help` lists every command and option.
|
|
141
|
+
|
|
142
|
+
## Status
|
|
143
|
+
|
|
144
|
+
Works with Claude Code. Agents sit behind a small adapter interface, so others
|
|
145
|
+
that support skills can be added.
|
|
146
|
+
|
|
147
|
+
## License
|
|
148
|
+
|
|
149
|
+
[MIT](LICENSE)
|
|
@@ -0,0 +1,384 @@
|
|
|
1
|
+
import { spawn } from "node:child_process";
|
|
2
|
+
import { mkdtemp, rm } from "node:fs/promises";
|
|
3
|
+
import * as os from "node:os";
|
|
4
|
+
import * as path from "node:path";
|
|
5
|
+
export const DEFAULT_DIRECTIVE =
|
|
6
|
+
// "Decide which skill fits" made sonnet answer with the name as text instead
|
|
7
|
+
// of calling the tool (0 of 6 runs); naming the Skill tool fixes it (5 of 6).
|
|
8
|
+
'[SKILL ROUTING CHECK] Invoke the Skill tool with the skill that fits this request. ' +
|
|
9
|
+
'Then stop immediately: no questions, no external sources, no file changes, no subagents. ' +
|
|
10
|
+
'If no skill fits, answer "none" and do not invoke any skill.';
|
|
11
|
+
function stringList(v) {
|
|
12
|
+
return Array.isArray(v) ? v.filter((s) => typeof s === "string") : null;
|
|
13
|
+
}
|
|
14
|
+
/**
|
|
15
|
+
* Incremental parser of claude's stream-json stdout. One event per line.
|
|
16
|
+
* Pushing lines also computes the early-stop signal, so the adapter can
|
|
17
|
+
* kill the process as soon as routing is decided.
|
|
18
|
+
*/
|
|
19
|
+
export class ClaudeStream {
|
|
20
|
+
loaded = [];
|
|
21
|
+
textParts = [];
|
|
22
|
+
resultText = "";
|
|
23
|
+
sawResult = false;
|
|
24
|
+
costUsd = null;
|
|
25
|
+
availableSkills = null;
|
|
26
|
+
initSkillList = null;
|
|
27
|
+
initSlashList = null;
|
|
28
|
+
sawInitEvent = false;
|
|
29
|
+
sawSkill = false;
|
|
30
|
+
structuredOutput = null;
|
|
31
|
+
failure = null;
|
|
32
|
+
/** Raw `skills` from the init event, null until init is parsed. */
|
|
33
|
+
get initSkills() {
|
|
34
|
+
return this.initSkillList;
|
|
35
|
+
}
|
|
36
|
+
/** Raw `slash_commands` from the init event, null until init is parsed. */
|
|
37
|
+
get initSlashCommands() {
|
|
38
|
+
return this.initSlashList;
|
|
39
|
+
}
|
|
40
|
+
/** True after the init event: the two lists above are then complete. */
|
|
41
|
+
get sawInit() {
|
|
42
|
+
return this.sawInitEvent;
|
|
43
|
+
}
|
|
44
|
+
/** Feed one stdout line; true means the run can be stopped early. */
|
|
45
|
+
push(line) {
|
|
46
|
+
if (!line.startsWith("{"))
|
|
47
|
+
return false;
|
|
48
|
+
let ev;
|
|
49
|
+
try {
|
|
50
|
+
ev = JSON.parse(line);
|
|
51
|
+
}
|
|
52
|
+
catch {
|
|
53
|
+
return false;
|
|
54
|
+
}
|
|
55
|
+
if (ev.type === "system" && ev.subtype === "init") {
|
|
56
|
+
this.sawInitEvent = true;
|
|
57
|
+
this.initSkillList = stringList(ev.skills);
|
|
58
|
+
this.initSlashList = stringList(ev.slash_commands);
|
|
59
|
+
// Commands (~/.claude/commands) are loadable through Skill too, but init
|
|
60
|
+
// lists them only in slash_commands, so take the union.
|
|
61
|
+
const names = [...(this.initSkillList ?? []), ...(this.initSlashList ?? [])];
|
|
62
|
+
if (this.initSkillList || this.initSlashList) {
|
|
63
|
+
this.availableSkills = [...new Set(names)];
|
|
64
|
+
}
|
|
65
|
+
return false;
|
|
66
|
+
}
|
|
67
|
+
if (ev.type === "assistant") {
|
|
68
|
+
const hadSkillBefore = this.sawSkill;
|
|
69
|
+
const hasSkill = this.handleAssistant(ev);
|
|
70
|
+
// Stop on the first assistant event without a Skill call that follows
|
|
71
|
+
// a Skill event; a second Skill keeps the run going (multi-skill cases).
|
|
72
|
+
return hadSkillBefore && !hasSkill;
|
|
73
|
+
}
|
|
74
|
+
if (ev.type === "result") {
|
|
75
|
+
this.sawResult = true;
|
|
76
|
+
this.resultText = typeof ev.result === "string" ? ev.result : "";
|
|
77
|
+
this.structuredOutput = ev.structured_output ?? null;
|
|
78
|
+
// error_max_turns is the normal end of a routing probe; any other error
|
|
79
|
+
// (usage limit, API failure, structured output retries exhausted) means
|
|
80
|
+
// the model never got to route.
|
|
81
|
+
if (ev.is_error === true && ev.subtype !== "error_max_turns") {
|
|
82
|
+
this.failure = `claude error: ${this.resultText.trim().slice(0, 120) || String(ev.subtype)}`;
|
|
83
|
+
}
|
|
84
|
+
if (typeof ev.total_cost_usd === "number")
|
|
85
|
+
this.costUsd = ev.total_cost_usd;
|
|
86
|
+
return true;
|
|
87
|
+
}
|
|
88
|
+
return false;
|
|
89
|
+
}
|
|
90
|
+
handleAssistant(ev) {
|
|
91
|
+
const message = ev.message;
|
|
92
|
+
const blocks = message?.content;
|
|
93
|
+
if (!Array.isArray(blocks))
|
|
94
|
+
return false;
|
|
95
|
+
let sawSkillHere = false;
|
|
96
|
+
for (const raw of blocks) {
|
|
97
|
+
if (typeof raw !== "object" || raw === null)
|
|
98
|
+
continue;
|
|
99
|
+
const block = raw;
|
|
100
|
+
if (block.type === "tool_use" && block.name === "Skill") {
|
|
101
|
+
const input = block.input;
|
|
102
|
+
const rawName = input?.skill ?? input?.command ?? input?.name;
|
|
103
|
+
const name = String(rawName ?? "").replace(/^\/+/, "");
|
|
104
|
+
if (name && !this.loaded.includes(name))
|
|
105
|
+
this.loaded.push(name);
|
|
106
|
+
sawSkillHere = true;
|
|
107
|
+
}
|
|
108
|
+
else if (block.type === "text" && typeof block.text === "string") {
|
|
109
|
+
this.textParts.push(block.text);
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
if (sawSkillHere)
|
|
113
|
+
this.sawSkill = true;
|
|
114
|
+
return sawSkillHere;
|
|
115
|
+
}
|
|
116
|
+
/** Error reported by claude itself in the result event, null if none. */
|
|
117
|
+
get error() {
|
|
118
|
+
return this.failure;
|
|
119
|
+
}
|
|
120
|
+
get result() {
|
|
121
|
+
const text = this.sawResult && this.resultText !== ""
|
|
122
|
+
? this.resultText
|
|
123
|
+
: this.textParts.join("\n");
|
|
124
|
+
return {
|
|
125
|
+
loaded: this.loaded,
|
|
126
|
+
text,
|
|
127
|
+
costUsd: this.costUsd,
|
|
128
|
+
availableSkills: this.availableSkills,
|
|
129
|
+
structuredOutput: this.structuredOutput,
|
|
130
|
+
};
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
export function parseClaudeStream(stdout) {
|
|
134
|
+
const stream = new ClaudeStream();
|
|
135
|
+
for (const line of stdout.split("\n"))
|
|
136
|
+
stream.push(line); // stop signal ignored here
|
|
137
|
+
return stream.result;
|
|
138
|
+
}
|
|
139
|
+
const DISALLOWED_TOOLS = [
|
|
140
|
+
"Bash", "Edit", "Write", "MultiEdit", "NotebookEdit",
|
|
141
|
+
"Agent", "Task", "WebFetch", "WebSearch", "mcp__*",
|
|
142
|
+
];
|
|
143
|
+
const STDERR_CAP = 4096;
|
|
144
|
+
const KILL_GRACE_MS = 2000;
|
|
145
|
+
function claudeArgs(opts) {
|
|
146
|
+
const args = [
|
|
147
|
+
"-p", `${opts.query}\n\n${opts.directive}`,
|
|
148
|
+
"--output-format", "stream-json",
|
|
149
|
+
"--verbose",
|
|
150
|
+
"--max-turns", "2",
|
|
151
|
+
"--permission-mode", "plan",
|
|
152
|
+
"--allowedTools", "Skill",
|
|
153
|
+
"--disallowedTools", ...DISALLOWED_TOOLS,
|
|
154
|
+
];
|
|
155
|
+
if (opts.model)
|
|
156
|
+
args.push("--model", opts.model);
|
|
157
|
+
return args;
|
|
158
|
+
}
|
|
159
|
+
function batchArgs(opts) {
|
|
160
|
+
// Skill is NOT disallowed here: removing the tool may also remove the skill
|
|
161
|
+
// list the model has to reason about; the survey is enforced by killing the
|
|
162
|
+
// process the moment a Skill call appears instead.
|
|
163
|
+
const args = [
|
|
164
|
+
"-p", opts.prompt,
|
|
165
|
+
"--output-format", "stream-json",
|
|
166
|
+
"--verbose",
|
|
167
|
+
"--max-turns", "3",
|
|
168
|
+
"--permission-mode", "plan",
|
|
169
|
+
"--json-schema", JSON.stringify(opts.schema),
|
|
170
|
+
"--disallowedTools", ...DISALLOWED_TOOLS,
|
|
171
|
+
];
|
|
172
|
+
if (opts.model)
|
|
173
|
+
args.push("--model", opts.model);
|
|
174
|
+
return args;
|
|
175
|
+
}
|
|
176
|
+
function listArgs() {
|
|
177
|
+
// The init event arrives before any model call, so the process is killed
|
|
178
|
+
// before anything is billed; the disallow list is a guard for that race.
|
|
179
|
+
return [
|
|
180
|
+
"-p", "none",
|
|
181
|
+
"--output-format", "stream-json",
|
|
182
|
+
"--verbose",
|
|
183
|
+
"--max-turns", "1",
|
|
184
|
+
"--permission-mode", "plan",
|
|
185
|
+
"--disallowedTools", ...DISALLOWED_TOOLS,
|
|
186
|
+
];
|
|
187
|
+
}
|
|
188
|
+
/** Seconds as written in the timeout message: 180000 -> "180", 400 -> "0.4". */
|
|
189
|
+
function seconds(ms) {
|
|
190
|
+
return String(Math.round((ms / 1000) * 100) / 100);
|
|
191
|
+
}
|
|
192
|
+
function spawnErrorMessage(e, bin) {
|
|
193
|
+
const err = e;
|
|
194
|
+
if (err.code === "ENOENT")
|
|
195
|
+
return `${bin} not found`;
|
|
196
|
+
return `spawn failed: ${err.message}`;
|
|
197
|
+
}
|
|
198
|
+
const LOGIN_HINT = " — with --config-dir, auth comes from ANTHROPIC_API_KEY or CLAUDE_CODE_OAUTH_TOKEN (claude setup-token)";
|
|
199
|
+
/** A fresh config dir has no credentials; point the user at the env vars. */
|
|
200
|
+
function withLoginHint(error, configDir) {
|
|
201
|
+
if (error && configDir && error.includes("Not logged in"))
|
|
202
|
+
return error + LOGIN_HINT;
|
|
203
|
+
return error;
|
|
204
|
+
}
|
|
205
|
+
/**
|
|
206
|
+
* Runs `claude -p` in a throwaway directory and reads its stream-json stdout
|
|
207
|
+
* incrementally, so the process can be killed as soon as routing is decided.
|
|
208
|
+
*/
|
|
209
|
+
export class ClaudeAdapter {
|
|
210
|
+
name = "claude";
|
|
211
|
+
async run(opts) {
|
|
212
|
+
const bin = process.env.SKILLCHECK_CLAUDE_BIN || "claude";
|
|
213
|
+
const workdir = await mkdtemp(path.join(os.tmpdir(), "skillcheck-"));
|
|
214
|
+
try {
|
|
215
|
+
const out = await this.streamRun(bin, claudeArgs(opts), workdir, opts.timeoutMs, {
|
|
216
|
+
earlyStop: opts.earlyStop,
|
|
217
|
+
configDir: opts.configDir,
|
|
218
|
+
});
|
|
219
|
+
const { loaded, text, costUsd, availableSkills } = out.stream.result;
|
|
220
|
+
const error = withLoginHint(out.error, opts.configDir);
|
|
221
|
+
return { loaded, text, costUsd, availableSkills, error, stoppedEarly: out.stoppedEarly, durationMs: out.durationMs };
|
|
222
|
+
}
|
|
223
|
+
finally {
|
|
224
|
+
await rm(workdir, { recursive: true, force: true });
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
async runBatch(opts) {
|
|
228
|
+
const bin = process.env.SKILLCHECK_CLAUDE_BIN || "claude";
|
|
229
|
+
const workdir = await mkdtemp(path.join(os.tmpdir(), "skillcheck-"));
|
|
230
|
+
try {
|
|
231
|
+
const out = await this.streamRun(bin, batchArgs(opts), workdir, opts.timeoutMs, {
|
|
232
|
+
earlyStop: false, // the run ends with the result event
|
|
233
|
+
configDir: opts.configDir,
|
|
234
|
+
abandon: (stream) => {
|
|
235
|
+
const { loaded } = stream.result;
|
|
236
|
+
const skill = loaded[loaded.length - 1];
|
|
237
|
+
return skill !== undefined ? `model loaded skill ${skill} instead of answering the survey` : null;
|
|
238
|
+
},
|
|
239
|
+
});
|
|
240
|
+
const { text, costUsd, structuredOutput } = out.stream.result;
|
|
241
|
+
return { structured: structuredOutput, text, costUsd, error: withLoginHint(out.error, opts.configDir), durationMs: out.durationMs };
|
|
242
|
+
}
|
|
243
|
+
finally {
|
|
244
|
+
await rm(workdir, { recursive: true, force: true });
|
|
245
|
+
}
|
|
246
|
+
}
|
|
247
|
+
/** Ask the agent for its skill lists; killed right after init, so no model call. */
|
|
248
|
+
async listSkills(opts) {
|
|
249
|
+
const bin = process.env.SKILLCHECK_CLAUDE_BIN || "claude";
|
|
250
|
+
const workdir = await mkdtemp(path.join(os.tmpdir(), "skillcheck-"));
|
|
251
|
+
try {
|
|
252
|
+
const out = await this.streamRun(bin, listArgs(), workdir, opts.timeoutMs, {
|
|
253
|
+
earlyStop: false,
|
|
254
|
+
configDir: opts.configDir,
|
|
255
|
+
stop: (stream) => stream.sawInit,
|
|
256
|
+
});
|
|
257
|
+
if (out.error || !out.stream.sawInit) {
|
|
258
|
+
return { skills: [], slashCommands: [], error: out.error ?? "claude sent no init event" };
|
|
259
|
+
}
|
|
260
|
+
const skills = [...new Set(out.stream.initSkills ?? [])].sort();
|
|
261
|
+
const skillSet = new Set(skills);
|
|
262
|
+
const slashCommands = [...new Set((out.stream.initSlashCommands ?? []).filter((s) => !skillSet.has(s)))].sort();
|
|
263
|
+
return { skills, slashCommands, error: null };
|
|
264
|
+
}
|
|
265
|
+
finally {
|
|
266
|
+
await rm(workdir, { recursive: true, force: true });
|
|
267
|
+
}
|
|
268
|
+
}
|
|
269
|
+
/**
|
|
270
|
+
* Shared spawn-and-parse loop. `earlyStop` enables the stream's stop signal;
|
|
271
|
+
* `abandon` runs after every event and, when it returns a message, the group
|
|
272
|
+
* is killed at once and that message becomes the run error. `stop` kills the
|
|
273
|
+
* group the same way but is not an error (used to end right after init).
|
|
274
|
+
*/
|
|
275
|
+
streamRun(bin, args, workdir, timeoutMs, control) {
|
|
276
|
+
return new Promise((resolve) => {
|
|
277
|
+
const startedAt = Date.now();
|
|
278
|
+
const stream = new ClaudeStream();
|
|
279
|
+
const env = { ...process.env };
|
|
280
|
+
delete env.CLAUDE_PROJECT_DIR; // the probe must not see the caller's project
|
|
281
|
+
if (control.configDir)
|
|
282
|
+
env.CLAUDE_CONFIG_DIR = control.configDir;
|
|
283
|
+
let child;
|
|
284
|
+
let tail = "";
|
|
285
|
+
let stdoutSeen = false;
|
|
286
|
+
let stderrText = "";
|
|
287
|
+
let error = null;
|
|
288
|
+
let stoppedEarly = false;
|
|
289
|
+
let settled = false;
|
|
290
|
+
let killTimer;
|
|
291
|
+
let timeoutTimer;
|
|
292
|
+
// detached => child.pid is the process group id; ESRCH means it is gone
|
|
293
|
+
const killGroup = (pid) => {
|
|
294
|
+
try {
|
|
295
|
+
process.kill(-pid, "SIGTERM");
|
|
296
|
+
}
|
|
297
|
+
catch {
|
|
298
|
+
// already exited
|
|
299
|
+
}
|
|
300
|
+
killTimer = setTimeout(() => {
|
|
301
|
+
try {
|
|
302
|
+
process.kill(-pid, "SIGKILL");
|
|
303
|
+
}
|
|
304
|
+
catch {
|
|
305
|
+
// already exited
|
|
306
|
+
}
|
|
307
|
+
}, KILL_GRACE_MS);
|
|
308
|
+
killTimer.unref();
|
|
309
|
+
};
|
|
310
|
+
const finish = (code, signal) => {
|
|
311
|
+
if (settled)
|
|
312
|
+
return;
|
|
313
|
+
settled = true;
|
|
314
|
+
if (killTimer)
|
|
315
|
+
clearTimeout(killTimer);
|
|
316
|
+
if (timeoutTimer)
|
|
317
|
+
clearTimeout(timeoutTimer);
|
|
318
|
+
// A non-zero exit is not a failure by itself: error_max_turns is normal.
|
|
319
|
+
// Only a stdout with nothing in it means the run really did not happen.
|
|
320
|
+
if (!error)
|
|
321
|
+
error = stream.error;
|
|
322
|
+
if (!error && !stdoutSeen) {
|
|
323
|
+
const detail = stderrText.trim().slice(0, 120);
|
|
324
|
+
error = `claude exited with code ${code ?? signal}${detail ? `: ${detail}` : ""}`;
|
|
325
|
+
}
|
|
326
|
+
resolve({ stream, error, stoppedEarly, durationMs: Date.now() - startedAt });
|
|
327
|
+
};
|
|
328
|
+
try {
|
|
329
|
+
child = spawn(bin, args, { cwd: workdir, env, stdio: ["ignore", "pipe", "pipe"], detached: true });
|
|
330
|
+
}
|
|
331
|
+
catch (e) {
|
|
332
|
+
error = spawnErrorMessage(e, bin);
|
|
333
|
+
finish(null, null);
|
|
334
|
+
return;
|
|
335
|
+
}
|
|
336
|
+
const proc = child;
|
|
337
|
+
proc.on("error", (e) => {
|
|
338
|
+
if (!error)
|
|
339
|
+
error = spawnErrorMessage(e, bin);
|
|
340
|
+
});
|
|
341
|
+
proc.stdout?.setEncoding("utf8");
|
|
342
|
+
proc.stdout?.on("data", (chunk) => {
|
|
343
|
+
if (!stdoutSeen && chunk.trim() !== "")
|
|
344
|
+
stdoutSeen = true;
|
|
345
|
+
tail += chunk;
|
|
346
|
+
let nl;
|
|
347
|
+
while ((nl = tail.indexOf("\n")) >= 0) {
|
|
348
|
+
const line = tail.slice(0, nl);
|
|
349
|
+
tail = tail.slice(nl + 1);
|
|
350
|
+
const stop = stream.push(line); // always parse: stop and abandon only gate the kill
|
|
351
|
+
if (error)
|
|
352
|
+
continue; // first failure wins, the group is already being killed
|
|
353
|
+
const abandoned = control.abandon?.(stream);
|
|
354
|
+
if (abandoned) {
|
|
355
|
+
error = abandoned;
|
|
356
|
+
if (proc.pid)
|
|
357
|
+
killGroup(proc.pid);
|
|
358
|
+
}
|
|
359
|
+
else if (!stoppedEarly && ((control.earlyStop && stop) || control.stop?.(stream) === true)) {
|
|
360
|
+
stoppedEarly = true;
|
|
361
|
+
if (proc.pid)
|
|
362
|
+
killGroup(proc.pid);
|
|
363
|
+
}
|
|
364
|
+
}
|
|
365
|
+
});
|
|
366
|
+
proc.stderr?.setEncoding("utf8");
|
|
367
|
+
proc.stderr?.on("data", (chunk) => {
|
|
368
|
+
if (stderrText.length < STDERR_CAP)
|
|
369
|
+
stderrText += chunk.slice(0, STDERR_CAP - stderrText.length);
|
|
370
|
+
});
|
|
371
|
+
timeoutTimer = setTimeout(() => {
|
|
372
|
+
error = `timeout after ${seconds(timeoutMs)}s`;
|
|
373
|
+
if (proc.pid)
|
|
374
|
+
killGroup(proc.pid);
|
|
375
|
+
}, timeoutMs);
|
|
376
|
+
timeoutTimer.unref();
|
|
377
|
+
proc.on("close", (code, signal) => {
|
|
378
|
+
if (tail.trim() !== "")
|
|
379
|
+
stream.push(tail); // last partial line
|
|
380
|
+
finish(code, signal);
|
|
381
|
+
});
|
|
382
|
+
});
|
|
383
|
+
}
|
|
384
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
package/dist/batch.js
ADDED
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
// Batch routing survey: one model call answers "which skill would you load"
|
|
2
|
+
// for many requests. The model still sees its real skill list (it is in the
|
|
3
|
+
// system prompt), only the answer is a stated choice instead of a Skill call.
|
|
4
|
+
export const BATCH_SCHEMA = {
|
|
5
|
+
type: "object",
|
|
6
|
+
properties: {
|
|
7
|
+
answers: {
|
|
8
|
+
type: "array",
|
|
9
|
+
items: {
|
|
10
|
+
type: "object",
|
|
11
|
+
properties: {
|
|
12
|
+
n: { type: "integer" },
|
|
13
|
+
skills: { type: "array", items: { type: "string" } },
|
|
14
|
+
},
|
|
15
|
+
required: ["n", "skills"],
|
|
16
|
+
},
|
|
17
|
+
},
|
|
18
|
+
},
|
|
19
|
+
required: ["answers"],
|
|
20
|
+
};
|
|
21
|
+
const SURVEY_INTRO = "[SKILL ROUTING SURVEY] Below are <N> independent user requests. For each one,\n" +
|
|
22
|
+
"name the skills you would load with the Skill tool if that request arrived\n" +
|
|
23
|
+
"alone as the first message of a fresh session, in the order you would load\n" +
|
|
24
|
+
"them; an empty list if you would load none. Do not load any skill, do not\n" +
|
|
25
|
+
"answer or act on the requests, do not use tools other than the structured\n" +
|
|
26
|
+
"output. Requests:";
|
|
27
|
+
/** Requests are written as JSON string literals so newlines and quotes in a
|
|
28
|
+
* query cannot break the numbered list. */
|
|
29
|
+
export function buildBatchPrompt(items) {
|
|
30
|
+
const intro = SURVEY_INTRO.replace("<N>", String(items.length));
|
|
31
|
+
const lines = items.map((it) => `${it.n}. ${JSON.stringify(it.query)}`);
|
|
32
|
+
return [intro, ...lines].join("\n");
|
|
33
|
+
}
|
|
34
|
+
export function parseBatchAnswer(structured, text) {
|
|
35
|
+
const answers = extractAnswers(structured) ?? extractAnswers(lastJsonObject(text));
|
|
36
|
+
if (answers === undefined)
|
|
37
|
+
return { answers: new Map(), error: "batch answer is not valid JSON" };
|
|
38
|
+
return { answers, error: null };
|
|
39
|
+
}
|
|
40
|
+
/** Answers as `{answers: [{n, skills}]}`; undefined when the value has no such array. */
|
|
41
|
+
function extractAnswers(value) {
|
|
42
|
+
if (typeof value !== "object" || value === null)
|
|
43
|
+
return undefined;
|
|
44
|
+
const answers = value.answers;
|
|
45
|
+
if (!Array.isArray(answers))
|
|
46
|
+
return undefined;
|
|
47
|
+
const out = new Map();
|
|
48
|
+
for (const raw of answers) {
|
|
49
|
+
if (typeof raw !== "object" || raw === null)
|
|
50
|
+
continue;
|
|
51
|
+
const entry = raw;
|
|
52
|
+
if (typeof entry.n !== "number" || !Number.isInteger(entry.n))
|
|
53
|
+
continue; // unnumberable entry
|
|
54
|
+
out.set(entry.n, cleanSkills(entry.skills));
|
|
55
|
+
}
|
|
56
|
+
return out;
|
|
57
|
+
}
|
|
58
|
+
function cleanSkills(value) {
|
|
59
|
+
if (!Array.isArray(value))
|
|
60
|
+
return [];
|
|
61
|
+
const out = [];
|
|
62
|
+
for (const raw of value) {
|
|
63
|
+
if (typeof raw !== "string")
|
|
64
|
+
continue;
|
|
65
|
+
const name = raw.trim().replace(/^\/+/, "");
|
|
66
|
+
if (name === "" || name === "none")
|
|
67
|
+
continue;
|
|
68
|
+
if (!out.includes(name))
|
|
69
|
+
out.push(name);
|
|
70
|
+
}
|
|
71
|
+
return out;
|
|
72
|
+
}
|
|
73
|
+
/** The last balanced top-level {…} in free text, braces inside JSON strings
|
|
74
|
+
* excluded from the scan; undefined when there is none or it does not parse. */
|
|
75
|
+
function lastJsonObject(text) {
|
|
76
|
+
let depth = 0;
|
|
77
|
+
let start = -1;
|
|
78
|
+
let inString = false;
|
|
79
|
+
let escaped = false;
|
|
80
|
+
let last;
|
|
81
|
+
for (let i = 0; i < text.length; i++) {
|
|
82
|
+
const ch = text[i];
|
|
83
|
+
if (inString) {
|
|
84
|
+
if (escaped)
|
|
85
|
+
escaped = false;
|
|
86
|
+
else if (ch === "\\")
|
|
87
|
+
escaped = true;
|
|
88
|
+
else if (ch === '"')
|
|
89
|
+
inString = false;
|
|
90
|
+
continue;
|
|
91
|
+
}
|
|
92
|
+
if (ch === '"')
|
|
93
|
+
inString = true;
|
|
94
|
+
else if (ch === "{") {
|
|
95
|
+
if (depth === 0)
|
|
96
|
+
start = i;
|
|
97
|
+
depth++;
|
|
98
|
+
}
|
|
99
|
+
else if (ch === "}" && depth > 0 && --depth === 0) {
|
|
100
|
+
last = text.slice(start, i + 1);
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
if (last === undefined)
|
|
104
|
+
return undefined;
|
|
105
|
+
try {
|
|
106
|
+
return JSON.parse(last);
|
|
107
|
+
}
|
|
108
|
+
catch {
|
|
109
|
+
return undefined;
|
|
110
|
+
}
|
|
111
|
+
}
|