@icntswm/skillcheck 1.1.0 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -120,6 +120,10 @@ skillcheck run --only 2,5 # confirm what the batch flagged
120
120
  `skillcheck list` prints the skills and slash commands Claude Code sees. It
121
121
  stops Claude Code before the first model call, so it costs nothing.
122
122
 
123
+ Already tuned a description with Anthropic's skill-creator? `skillcheck import
124
+ eval_set.json --skill <name>` turns its trigger eval set into cases (see
125
+ [Writing cases](docs/writing-cases.md#from-skill-creator)).
126
+
123
127
  ## Documentation
124
128
 
125
129
  | | |
@@ -327,7 +327,8 @@ export class ClaudeAdapter {
327
327
  },
328
328
  });
329
329
  const { text, costUsd, structuredOutput } = out.stream.result;
330
- return { structured: structuredOutput, text, costUsd, error: withLoginHint(out.error, opts.configDir), durationMs: out.durationMs };
330
+ const error = withLoginHint(out.error, opts.configDir);
331
+ return { structured: structuredOutput, text, costUsd, usage: out.stream.usage, error, durationMs: out.durationMs };
331
332
  }
332
333
  finally {
333
334
  await rm(workdir, { recursive: true, force: true });
@@ -408,13 +409,15 @@ export class ClaudeAdapter {
408
409
  clearTimeout(killTimer);
409
410
  if (timeoutTimer)
410
411
  clearTimeout(timeoutTimer);
411
- // A non-zero exit is not a failure by itself: error_max_turns is normal.
412
- // Only a stdout with nothing in it means the run really did not happen.
412
+ // A non-zero exit is not a failure by itself: error_max_turns is normal
413
+ // and ends with a result event. Without one, the run ended before routing
414
+ // was over, unless we stopped it ourselves.
413
415
  if (!error)
414
416
  error = stream.error;
415
- if (!error && !stdoutSeen) {
417
+ if (!error && !stopping && !stream.finished) {
416
418
  const detail = stderrText.trim().slice(0, 120);
417
- error = `claude exited with code ${code ?? signal}${detail ? `: ${detail}` : ""}`;
419
+ const why = stdoutSeen ? `claude exited with code ${code ?? signal} before its result` : `claude exited with code ${code ?? signal}`;
420
+ error = `${why}${detail ? `: ${detail}` : ""}`;
418
421
  }
419
422
  resolve({ stream, error, stoppedEarly, durationMs: Date.now() - startedAt });
420
423
  };
package/dist/cli.js CHANGED
@@ -11,6 +11,7 @@ import { BUILTIN_SKILLS, ConfigError, findDefaultCasesFile, knownSkillNames, loa
11
11
  import { confusion } from "./confusion.js";
12
12
  import { loadSkillDocs } from "./describe.js";
13
13
  import { aggregate, judge } from "./judge.js";
14
+ import { evalSetToSuite } from "./import.js";
14
15
  import { toJunit } from "./junit.js";
15
16
  import { lint, LINT_DEFAULTS } from "./lint.js";
16
17
  import { toMarkdown } from "./markdown.js";
@@ -26,6 +27,8 @@ Usage:
26
27
  skillcheck lint [file] [options] static checks on descriptions and cases
27
28
  skillcheck list [options] print the skills and commands the agent can load
28
29
  skillcheck init [file] [options] write a starter cases file with those names
30
+ skillcheck import <file> --skill <name>
31
+ turn a skill-creator trigger eval set into cases
29
32
 
30
33
  run options:
31
34
  -a, --agent <name> agent to route with (default: suite or claude)
@@ -59,6 +62,9 @@ list/init options:
59
62
  CLAUDE_CONFIG_DIR; run, list and init also set it for the agent
60
63
  init options:
61
64
  --force overwrite an existing file
65
+ import options:
66
+ --skill <name> the skill the eval set is about
67
+ -o, --out <file> write the cases there instead of stdout (--force overwrites)
62
68
  common:
63
69
  -h, --help show this help
64
70
  --version show version
@@ -69,6 +75,7 @@ nothing reaches the model and nothing is billed, even when not logged in.
69
75
 
70
76
  Exit codes: 0 all passed, 1 some case failed or was skipped, 2 config or environment error.
71
77
  `;
78
+ const COMMANDS = ["run", "check", "lint", "list", "init", "import"];
72
79
  const OPTIONS = {
73
80
  help: { type: "boolean", short: "h", default: false },
74
81
  version: { type: "boolean", default: false },
@@ -94,6 +101,7 @@ const OPTIONS = {
94
101
  strict: { type: "boolean", default: false },
95
102
  "config-dir": { type: "string" },
96
103
  force: { type: "boolean", default: false },
104
+ out: { type: "string", short: "o" },
97
105
  };
98
106
  /** Config/environment problems: message to stderr, exit 2. */
99
107
  class UsageError extends Error {
@@ -129,6 +137,7 @@ function parseFlags(values, cwd) {
129
137
  strict: values.strict,
130
138
  configDir: values["config-dir"] !== undefined ? resolveConfigDir(values["config-dir"], cwd) : undefined,
131
139
  force: values.force,
140
+ out: values.out,
132
141
  };
133
142
  }
134
143
  /** --config-dir: absolute, must be an existing directory. */
@@ -203,7 +212,7 @@ export async function main(argv, io = { stdout: process.stdout, stderr: process.
203
212
  io.stderr.write(USAGE);
204
213
  return 2;
205
214
  }
206
- if (command !== "run" && command !== "check" && command !== "lint" && command !== "list" && command !== "init") {
215
+ if (!COMMANDS.includes(command)) {
207
216
  io.stderr.write(`skillcheck: unknown command "${command}"\n\n${USAGE}`);
208
217
  return 2;
209
218
  }
@@ -214,6 +223,8 @@ export async function main(argv, io = { stdout: process.stdout, stderr: process.
214
223
  return await listCommand(flags, io, deps);
215
224
  if (command === "init")
216
225
  return await initCommand(positionals[1], flags, io, deps);
226
+ if (command === "import")
227
+ return importCommand(positionals[1], flags, io);
217
228
  const file = casesFile(positionals[1], io);
218
229
  if (command === "run")
219
230
  return await runCommand(file, flags, io, deps);
@@ -376,6 +387,47 @@ async function initCommand(positional, flags, io, deps) {
376
387
  io.stdout.write(`wrote ${file} (${skills.length} skills listed)\n`);
377
388
  return 0;
378
389
  }
390
+ function importCommand(positional, flags, io) {
391
+ if (positional === undefined) {
392
+ throw new UsageError("import needs the eval set file: skillcheck import <eval_set.json> --skill <name>");
393
+ }
394
+ const skill = flags.skill?.trim();
395
+ if (!skill || skill.includes(","))
396
+ throw new UsageError("import needs --skill <name>: the skill the eval set is about");
397
+ const source = path.resolve(io.cwd, positional);
398
+ let data;
399
+ try {
400
+ data = JSON.parse(fs.readFileSync(source, "utf8"));
401
+ }
402
+ catch (e) {
403
+ throw new UsageError(`cannot read ${positional}: ${e.message}`);
404
+ }
405
+ let suite;
406
+ try {
407
+ suite = evalSetToSuite(data, skill, source);
408
+ }
409
+ catch (e) {
410
+ if (e instanceof ConfigError)
411
+ throw new UsageError(`${positional}: ${e.errors.join("; ")}`);
412
+ throw e;
413
+ }
414
+ if (flags.out === undefined) {
415
+ io.stdout.write(suite.text);
416
+ return 0;
417
+ }
418
+ const out = path.resolve(io.cwd, flags.out);
419
+ if (fs.existsSync(out) && !flags.force)
420
+ throw new UsageError(`${flags.out} exists, use --force to overwrite`);
421
+ try {
422
+ fs.writeFileSync(out, suite.text);
423
+ }
424
+ catch (e) {
425
+ throw new UsageError(`cannot write ${flags.out}: ${e.message}`);
426
+ }
427
+ const total = suite.positive + suite.negative;
428
+ io.stdout.write(`wrote ${flags.out} (${total} cases: ${suite.positive} should load ${skill}, ${suite.negative} should not)\n`);
429
+ return 0;
430
+ }
379
431
  const UNCOVERED_WIDTH = 78;
380
432
  function printLint(out, report, docs, suite) {
381
433
  const casesPart = suite ? `${report.cases} cases` : "no cases file";
@@ -477,8 +529,8 @@ async function runCommand(file, flags, io, deps) {
477
529
  const startedAtMs = Date.now();
478
530
  const fatal = new FatalStop();
479
531
  const outcome = flags.batch
480
- ? await runBatched(adapter.runBatch.bind(adapter), items, flags, model, repeat, reporter, budget, fatal)
481
- : await runIndividually(adapter, items, flags, model, directive, repeat, reporter, budget, fatal);
532
+ ? await runBatched(adapter.runBatch.bind(adapter), items, flags, model, reporter, budget, fatal)
533
+ : await runIndividually(adapter, items, flags, model, directive, reporter, budget, fatal);
482
534
  if (fatal.message !== null) {
483
535
  // an environment problem, not a routing result: no summary, no reports
484
536
  io.stderr.write(`skillcheck: stopped, the agent cannot run: ${fatal.message}\n`);
@@ -489,6 +541,7 @@ async function runCommand(file, flags, io, deps) {
489
541
  for (const item of skipped)
490
542
  reporter.caseSkipped(item.c);
491
543
  const notStartedRuns = skipped.reduce((n, item) => n + item.runs.length - item.done, 0);
544
+ const skippedRunCosts = skipped.flatMap((item) => item.runs.flatMap((v) => (v ? [v.costUsd] : [])));
492
545
  const ordered = outcome.results.filter((r) => r !== undefined);
493
546
  const expected = expectedNames(selected);
494
547
  const unavailable = outcome.sawAvailability ? expected.filter((name) => !outcome.available.has(name)) : [];
@@ -499,6 +552,7 @@ async function runCommand(file, flags, io, deps) {
499
552
  skipped: skipped.length,
500
553
  budget: skipped.length > 0 ? { limitUsd: flags.budget, spent: budget.spent, notStartedRuns } : undefined,
501
554
  estimatedUsd: budget.spent,
555
+ skippedRunCosts,
502
556
  });
503
557
  if (flags.batch) {
504
558
  // cases that only errored have no answer to confirm
@@ -519,6 +573,7 @@ async function runCommand(file, flags, io, deps) {
519
573
  confusion: pairs,
520
574
  batch: flags.batch,
521
575
  estimatedCostUsd: budget.spent,
576
+ skippedRunCosts,
522
577
  budgetUsd: flags.budget ?? null,
523
578
  budgetReached: skipped.length > 0,
524
579
  });
@@ -544,9 +599,9 @@ function writeReportFile(path, text, flag) {
544
599
  }
545
600
  }
546
601
  /** One agent call per case: the normal mode. */
547
- async function runIndividually(adapter, items, flags, model, directive, repeat, reporter, budget, fatal) {
602
+ async function runIndividually(adapter, items, flags, model, directive, reporter, budget, fatal) {
548
603
  const totalRuns = items.reduce((n, item) => n + item.runs.length, 0);
549
- reporter.header(items.length, repeat, 1, totalRuns);
604
+ reporter.header(items.length, repeatLabel(items), 1, totalRuns);
550
605
  const jobs = items.flatMap((item, jobIndex) => item.runs.map((_, runIndex) => ({ item, runIndex, jobIndex })));
551
606
  // keep cases in file order even though runs of different cases interleave
552
607
  const results = new Array(items.length);
@@ -582,17 +637,19 @@ async function runIndividually(adapter, items, flags, model, directive, repeat,
582
637
  }
583
638
  /** One agent call per chunk of cases; the model states its choice per request
584
639
  * instead of loading skills. Verdicts flow through the same judge. */
585
- async function runBatched(runBatch, items, flags, model, repeat, reporter, budget, fatal) {
640
+ async function runBatched(runBatch, items, flags, model, reporter, budget, fatal) {
586
641
  const chunks = [];
587
642
  for (let i = 0; i < items.length; i += flags.batchSize)
588
643
  chunks.push(items.slice(i, i + flags.batchSize));
589
644
  const rounds = Math.max(...items.map((item) => item.runs.length));
590
645
  const jobs = [];
591
646
  for (let round = 0; round < rounds; round++) {
647
+ // a round past every repeat in the chunk makes no call
592
648
  for (const chunk of chunks)
593
- jobs.push({ chunk, round });
649
+ if (chunk.some((item) => round < item.runs.length))
650
+ jobs.push({ chunk, round });
594
651
  }
595
- reporter.batchHeader(items.length, repeat, chunks.length * rounds);
652
+ reporter.batchHeader(items.length, repeatLabel(items), jobs.length);
596
653
  const results = new Array(items.length);
597
654
  await runPool(jobs, flags.jobs, async (job) => {
598
655
  // a case with a smaller repeat only takes its first rounds
@@ -604,6 +661,8 @@ async function runBatched(runBatch, items, flags, model, repeat, reporter, budge
604
661
  const parsed = parseBatchAnswer(br.structured, br.text);
605
662
  const chunkError = br.error ?? parsed.error; // a chunk error is every case's error
606
663
  fatal.note(br.error);
664
+ // one call, one budget entry: its usage prices it when the cost never came
665
+ budget.add(br.costUsd, br.usage ?? null);
607
666
  active.forEach((item, i) => {
608
667
  const answer = parsed.answers.get(i + 1);
609
668
  const r = {
@@ -616,7 +675,6 @@ async function runBatched(runBatch, items, flags, model, repeat, reporter, budge
616
675
  durationMs: br.durationMs,
617
676
  };
618
677
  const verdict = judge(item.c, r);
619
- budget.add(verdict.costUsd);
620
678
  item.runs[job.round] = verdict;
621
679
  item.done++;
622
680
  if (item.done === item.runs.length) {
@@ -628,6 +686,13 @@ async function runBatched(runBatch, items, flags, model, repeat, reporter, budge
628
686
  }, { shouldStart: () => fatal.canStart(budget) });
629
687
  return { results, available: new Set(), sawAvailability: false };
630
688
  }
689
+ /** "3", or "1–3" when cases override the repeat. */
690
+ function repeatLabel(items) {
691
+ const counts = items.map((item) => item.runs.length);
692
+ const min = Math.min(...counts);
693
+ const max = Math.max(...counts);
694
+ return min === max ? String(min) : `${min}–${max}`;
695
+ }
631
696
  function selectCases(suite, agent, only) {
632
697
  const forAgent = suite.cases.filter((c) => !c.agents || c.agents.includes(agent));
633
698
  if (!only)
package/dist/import.js ADDED
@@ -0,0 +1,50 @@
1
+ import * as path from "node:path";
2
+ import { parse as parseYaml } from "yaml";
3
+ import { ConfigError, parseSuite } from "./cases.js";
4
+ const PLAIN_NAME = /^[A-Za-z0-9._:/-]+$/;
5
+ /**
6
+ * A skill-creator trigger eval set ([{query, should_trigger}], one skill) as a
7
+ * cases file: should_trigger true becomes expect, false becomes forbid.
8
+ */
9
+ export function evalSetToSuite(data, skill, source) {
10
+ const errors = [];
11
+ if (!Array.isArray(data) || data.length === 0) {
12
+ throw new ConfigError(["expected a JSON array of {query, should_trigger}"]);
13
+ }
14
+ const items = [];
15
+ data.forEach((item, i) => {
16
+ const n = i + 1;
17
+ if (typeof item !== "object" || item === null || Array.isArray(item)) {
18
+ errors.push(`item ${n}: expected an object with query and should_trigger`);
19
+ return;
20
+ }
21
+ const { query, should_trigger: trigger } = item;
22
+ if (typeof query !== "string" || query.trim() === "")
23
+ errors.push(`item ${n}: query must be a non-empty string`);
24
+ if (typeof trigger !== "boolean")
25
+ errors.push(`item ${n}: should_trigger must be true or false`);
26
+ if (typeof query === "string" && typeof trigger === "boolean")
27
+ items.push({ query, trigger });
28
+ });
29
+ if (errors.length > 0)
30
+ throw new ConfigError(errors);
31
+ // plain only when YAML reads it back as the same string: not true, null, 123
32
+ const name = PLAIN_NAME.test(skill) && parseYaml(skill) === skill ? skill : JSON.stringify(skill);
33
+ const lines = [
34
+ `# skillcheck cases imported from ${path.basename(source)} (skill-creator trigger eval set).`,
35
+ "# should_trigger: true became expect, false became forbid: another skill may",
36
+ `# still load there, only ${skill} must not.`,
37
+ "agent: claude",
38
+ "repeat: 1",
39
+ "threshold: 1.0",
40
+ "cases:",
41
+ ];
42
+ for (const { query, trigger } of items) {
43
+ // a JSON string is a valid YAML double-quoted scalar, newlines and quotes included
44
+ lines.push(` - query: ${JSON.stringify(query)}`, ` ${trigger ? "expect" : "forbid"}: [${name}]`);
45
+ }
46
+ const text = lines.join("\n") + "\n";
47
+ parseSuite(parseYaml(text));
48
+ const positive = items.filter((i) => i.trigger).length;
49
+ return { text, positive, negative: items.length - positive };
50
+ }
package/dist/markdown.js CHANGED
@@ -96,7 +96,7 @@ function failMark(c) {
96
96
  return bad.length > 0 && bad.every((r) => r.error !== null) ? "⚠️" : "❌";
97
97
  }
98
98
  /** The first non-ok run's reason, the pass share prefixed over several runs,
99
- * a diagnosis appended in italics. */
99
+ * a diagnosis appended in italics, then the case note. */
100
100
  function reasonCell(c) {
101
101
  const bad = c.runs.filter((r) => !r.ok);
102
102
  const parts = c.runs.length > 1 ? [`${c.passed}/${c.runs.length}`, bad[0]?.reason ?? ""] : [bad[0]?.reason ?? ""];
@@ -104,6 +104,8 @@ function reasonCell(c) {
104
104
  const diagnosis = bad.find((r) => r.diagnosis !== null)?.diagnosis;
105
105
  if (diagnosis)
106
106
  text += ` — _${diagnosis}_`;
107
+ if (c.note !== null)
108
+ text += ` · note: ${c.note}`;
107
109
  return text;
108
110
  }
109
111
  function passedRow(c) {
package/dist/report.js CHANGED
@@ -11,7 +11,7 @@ export class Reporter {
11
11
  this.useColor = out.isTTY === true && !process.env.NO_COLOR;
12
12
  }
13
13
  header(nCases, repeat, nAgents, totalRuns) {
14
- const runs = totalRuns ?? nCases * repeat * nAgents;
14
+ const runs = totalRuns ?? nCases * Number(repeat) * nAgents;
15
15
  const agents = nAgents === 1 ? "agent" : "agents";
16
16
  this.write(`${nCases} cases × ${repeat} repeat × ${nAgents} ${agents} = ${runs} runs\n`);
17
17
  }
@@ -36,6 +36,8 @@ export class Reporter {
36
36
  if (diagnosed?.diagnosis) {
37
37
  this.write(` ${this.paint("33", `diagnosis ${label}: ${diagnosed.diagnosis}`)}\n`);
38
38
  }
39
+ if (!res.ok && res.case.note)
40
+ this.write(` ${this.paint("2", `note ${label}: ${oneLine(res.case.note)}`)}\n`);
39
41
  }
40
42
  /** A case the budget never let start; printed after all real runs settle. */
41
43
  caseSkipped(c) {
@@ -68,7 +70,8 @@ export class Reporter {
68
70
  const failed = results.filter((r) => !r.ok).length;
69
71
  const skipped = extra?.skipped ?? 0;
70
72
  const skippedPart = skipped > 0 ? `, ${skipped} skipped` : "";
71
- this.write(`${failed} failed${skippedPart} of ${results.length + skipped} · runs ${verdicts.length} · ${costLine(verdicts.map((v) => v.costUsd), extra?.estimatedUsd)}\n`);
73
+ const costs = [...verdicts.map((v) => v.costUsd), ...(extra?.skippedRunCosts ?? [])];
74
+ this.write(`${failed} failed${skippedPart} of ${results.length + skipped} · runs ${costs.length} · ${costLine(costs, extra?.estimatedUsd)}\n`);
72
75
  }
73
76
  paint(code, text) {
74
77
  return this.useColor ? `\u001b[${code}m${text}${RESET}` : text;
package/dist/results.js CHANGED
@@ -2,7 +2,7 @@ import { readVersion } from "./version.js";
2
2
  /** The single source of truth for --json and --junit; computed once per run. */
3
3
  export function buildReport(input) {
4
4
  const verdicts = input.cases.flatMap((e) => e.result?.runs ?? []);
5
- const costs = verdicts.map((v) => v.costUsd);
5
+ const costs = [...verdicts.map((v) => v.costUsd), ...(input.skippedRunCosts ?? [])];
6
6
  const known = costs.filter((c) => c !== null);
7
7
  return {
8
8
  tool: "skillcheck",
@@ -17,7 +17,7 @@ export function buildReport(input) {
17
17
  cases: input.cases.length,
18
18
  failed: input.cases.filter((e) => e.result && !e.result.ok).length,
19
19
  skipped: input.cases.filter((e) => !e.result).length,
20
- runs: verdicts.length,
20
+ runs: costs.length,
21
21
  costUsd: known.reduce((a, b) => a + b, 0),
22
22
  unknownCostRuns: costs.length - known.length,
23
23
  estimatedCostUsd: input.estimatedCostUsd,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@icntswm/skillcheck",
3
- "version": "1.1.0",
3
+ "version": "1.2.0",
4
4
  "description": "Regression tests for Claude Code skills: check that every request still loads the right skill",
5
5
  "type": "module",
6
6
  "bin": {