@icntswm/skillcheck 1.0.4 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -36,9 +36,10 @@ requests.
36
36
  from a broken one instead of letting you guess.
37
37
  - 🏷️ **Catches renames and typos.** Case names are checked against the skills
38
38
  the agent really has, so a renamed skill can't pass silently.
39
- - ⚙️ **CI-ready.** A GitHub Action (`uses: icntswm/skillcheck@v1`), JUnit
40
- and JSON reports, clear exit codes, a spending cap (`--budget`), and
41
- `--config-dir` to test only the skills in your repository.
39
+ - ⚙️ **CI-ready.** A GitHub Action (`uses: icntswm/skillcheck@v1`) that
40
+ comments the report on the pull request, JUnit, JSON and Markdown reports,
41
+ clear exit codes, a spending cap (`--budget`), and `--config-dir` to test
42
+ only the skills in your repository.
42
43
  - 📄 **Plain YAML, one dependency.** Cases are readable by anyone on the team
43
44
  and live next to the skills they test.
44
45
 
@@ -119,13 +120,17 @@ skillcheck run --only 2,5 # confirm what the batch flagged
119
120
  `skillcheck list` prints the skills and slash commands Claude Code sees. It
120
121
  stops Claude Code before the first model call, so it costs nothing.
121
122
 
123
+ Already tuned a description with Anthropic's skill-creator? `skillcheck import
124
+ eval_set.json --skill <name>` turns its trigger eval set into cases (see
125
+ [Writing cases](docs/writing-cases.md#from-skill-creator)).
126
+
122
127
  ## Documentation
123
128
 
124
129
  | | |
125
130
  |---|---|
126
131
  | [Writing cases](docs/writing-cases.md) | the file format and what makes a case catch regressions |
127
132
  | [Cost](docs/cost.md) | what a run costs and how to spend less |
128
- | [Reports and CI](docs/ci.md) | confusion block, JSON, JUnit, GitHub Actions, exit codes |
133
+ | [Reports and CI](docs/ci.md) | confusion block, JSON, JUnit, Markdown, GitHub Actions, exit codes |
129
134
  | [How it works](docs/how-it-works.md) | what happens inside a run, and the limits of each level |
130
135
 
131
136
  `skillcheck --help` lists every command and option.
@@ -327,7 +327,8 @@ export class ClaudeAdapter {
327
327
  },
328
328
  });
329
329
  const { text, costUsd, structuredOutput } = out.stream.result;
330
- return { structured: structuredOutput, text, costUsd, error: withLoginHint(out.error, opts.configDir), durationMs: out.durationMs };
330
+ const error = withLoginHint(out.error, opts.configDir);
331
+ return { structured: structuredOutput, text, costUsd, usage: out.stream.usage, error, durationMs: out.durationMs };
331
332
  }
332
333
  finally {
333
334
  await rm(workdir, { recursive: true, force: true });
@@ -408,13 +409,15 @@ export class ClaudeAdapter {
408
409
  clearTimeout(killTimer);
409
410
  if (timeoutTimer)
410
411
  clearTimeout(timeoutTimer);
411
- // A non-zero exit is not a failure by itself: error_max_turns is normal.
412
- // Only a stdout with nothing in it means the run really did not happen.
412
+ // A non-zero exit is not a failure by itself: error_max_turns is normal
413
+ // and ends with a result event. Without one, the run ended before routing
414
+ // was over, unless we stopped it ourselves.
413
415
  if (!error)
414
416
  error = stream.error;
415
- if (!error && !stdoutSeen) {
417
+ if (!error && !stopping && !stream.finished) {
416
418
  const detail = stderrText.trim().slice(0, 120);
417
- error = `claude exited with code ${code ?? signal}${detail ? `: ${detail}` : ""}`;
419
+ const why = stdoutSeen ? `claude exited with code ${code ?? signal} before its result` : `claude exited with code ${code ?? signal}`;
420
+ error = `${why}${detail ? `: ${detail}` : ""}`;
418
421
  }
419
422
  resolve({ stream, error, stoppedEarly, durationMs: Date.now() - startedAt });
420
423
  };
package/dist/cli.js CHANGED
@@ -11,8 +11,10 @@ import { BUILTIN_SKILLS, ConfigError, findDefaultCasesFile, knownSkillNames, loa
11
11
  import { confusion } from "./confusion.js";
12
12
  import { loadSkillDocs } from "./describe.js";
13
13
  import { aggregate, judge } from "./judge.js";
14
+ import { evalSetToSuite } from "./import.js";
14
15
  import { toJunit } from "./junit.js";
15
16
  import { lint, LINT_DEFAULTS } from "./lint.js";
17
+ import { toMarkdown } from "./markdown.js";
16
18
  import { runPool } from "./pool.js";
17
19
  import { oneLine, Reporter } from "./report.js";
18
20
  import { buildReport } from "./results.js";
@@ -25,6 +27,8 @@ Usage:
25
27
  skillcheck lint [file] [options] static checks on descriptions and cases
26
28
  skillcheck list [options] print the skills and commands the agent can load
27
29
  skillcheck init [file] [options] write a starter cases file with those names
30
+ skillcheck import <file> --skill <name>
31
+ turn a skill-creator trigger eval set into cases
28
32
 
29
33
  run options:
30
34
  -a, --agent <name> agent to route with (default: suite or claude)
@@ -42,6 +46,7 @@ run options:
42
46
  --json <path> machine-readable report to path ("-" writes it to stdout
43
47
  and the terminal report to stderr)
44
48
  --junit <path> JUnit XML report (GitLab/GitHub test reporters) to path
49
+ --markdown <path> Markdown summary (PR comments, GitHub job summary) to path
45
50
  run/check options:
46
51
  --skill <a,b> keep only cases that mention these skills
47
52
  check options:
@@ -57,6 +62,9 @@ list/init options:
57
62
  CLAUDE_CONFIG_DIR; run, list and init also set it for the agent
58
63
  init options:
59
64
  --force overwrite an existing file
65
+ import options:
66
+ --skill <name> the skill the eval set is about
67
+ -o, --out <file> write the cases there instead of stdout (--force overwrites)
60
68
  common:
61
69
  -h, --help show this help
62
70
  --version show version
@@ -67,6 +75,7 @@ nothing reaches the model and nothing is billed, even when not logged in.
67
75
 
68
76
  Exit codes: 0 all passed, 1 some case failed or was skipped, 2 config or environment error.
69
77
  `;
78
+ const COMMANDS = ["run", "check", "lint", "list", "init", "import"];
70
79
  const OPTIONS = {
71
80
  help: { type: "boolean", short: "h", default: false },
72
81
  version: { type: "boolean", default: false },
@@ -86,11 +95,13 @@ const OPTIONS = {
86
95
  budget: { type: "string" },
87
96
  json: { type: "string" },
88
97
  junit: { type: "string" },
98
+ markdown: { type: "string" },
89
99
  top: { type: "string" },
90
100
  overlap: { type: "string" },
91
101
  strict: { type: "boolean", default: false },
92
102
  "config-dir": { type: "string" },
93
103
  force: { type: "boolean", default: false },
104
+ out: { type: "string", short: "o" },
94
105
  };
95
106
  /** Config/environment problems: message to stderr, exit 2. */
96
107
  class UsageError extends Error {
@@ -120,11 +131,13 @@ function parseFlags(values, cwd) {
120
131
  budget: numberFlag(values.budget),
121
132
  json: values.json,
122
133
  junit: values.junit,
134
+ markdown: values.markdown,
123
135
  top: intFlag(values.top, "top", 1) ?? 5,
124
136
  overlap: ratioFlag(values.overlap) ?? 0.3,
125
137
  strict: values.strict,
126
138
  configDir: values["config-dir"] !== undefined ? resolveConfigDir(values["config-dir"], cwd) : undefined,
127
139
  force: values.force,
140
+ out: values.out,
128
141
  };
129
142
  }
130
143
  /** --config-dir: absolute, must be an existing directory. */
@@ -199,7 +212,7 @@ export async function main(argv, io = { stdout: process.stdout, stderr: process.
199
212
  io.stderr.write(USAGE);
200
213
  return 2;
201
214
  }
202
- if (command !== "run" && command !== "check" && command !== "lint" && command !== "list" && command !== "init") {
215
+ if (!COMMANDS.includes(command)) {
203
216
  io.stderr.write(`skillcheck: unknown command "${command}"\n\n${USAGE}`);
204
217
  return 2;
205
218
  }
@@ -210,6 +223,8 @@ export async function main(argv, io = { stdout: process.stdout, stderr: process.
210
223
  return await listCommand(flags, io, deps);
211
224
  if (command === "init")
212
225
  return await initCommand(positionals[1], flags, io, deps);
226
+ if (command === "import")
227
+ return importCommand(positionals[1], flags, io);
213
228
  const file = casesFile(positionals[1], io);
214
229
  if (command === "run")
215
230
  return await runCommand(file, flags, io, deps);
@@ -372,6 +387,47 @@ async function initCommand(positional, flags, io, deps) {
372
387
  io.stdout.write(`wrote ${file} (${skills.length} skills listed)\n`);
373
388
  return 0;
374
389
  }
390
+ function importCommand(positional, flags, io) {
391
+ if (positional === undefined) {
392
+ throw new UsageError("import needs the eval set file: skillcheck import <eval_set.json> --skill <name>");
393
+ }
394
+ const skill = flags.skill?.trim();
395
+ if (!skill || skill.includes(","))
396
+ throw new UsageError("import needs --skill <name>: the skill the eval set is about");
397
+ const source = path.resolve(io.cwd, positional);
398
+ let data;
399
+ try {
400
+ data = JSON.parse(fs.readFileSync(source, "utf8"));
401
+ }
402
+ catch (e) {
403
+ throw new UsageError(`cannot read ${positional}: ${e.message}`);
404
+ }
405
+ let suite;
406
+ try {
407
+ suite = evalSetToSuite(data, skill, source);
408
+ }
409
+ catch (e) {
410
+ if (e instanceof ConfigError)
411
+ throw new UsageError(`${positional}: ${e.errors.join("; ")}`);
412
+ throw e;
413
+ }
414
+ if (flags.out === undefined) {
415
+ io.stdout.write(suite.text);
416
+ return 0;
417
+ }
418
+ const out = path.resolve(io.cwd, flags.out);
419
+ if (fs.existsSync(out) && !flags.force)
420
+ throw new UsageError(`${flags.out} exists, use --force to overwrite`);
421
+ try {
422
+ fs.writeFileSync(out, suite.text);
423
+ }
424
+ catch (e) {
425
+ throw new UsageError(`cannot write ${flags.out}: ${e.message}`);
426
+ }
427
+ const total = suite.positive + suite.negative;
428
+ io.stdout.write(`wrote ${flags.out} (${total} cases: ${suite.positive} should load ${skill}, ${suite.negative} should not)\n`);
429
+ return 0;
430
+ }
375
431
  const UNCOVERED_WIDTH = 78;
376
432
  function printLint(out, report, docs, suite) {
377
433
  const casesPart = suite ? `${report.cases} cases` : "no cases file";
@@ -473,8 +529,8 @@ async function runCommand(file, flags, io, deps) {
473
529
  const startedAtMs = Date.now();
474
530
  const fatal = new FatalStop();
475
531
  const outcome = flags.batch
476
- ? await runBatched(adapter.runBatch.bind(adapter), items, flags, model, repeat, reporter, budget, fatal)
477
- : await runIndividually(adapter, items, flags, model, directive, repeat, reporter, budget, fatal);
532
+ ? await runBatched(adapter.runBatch.bind(adapter), items, flags, model, reporter, budget, fatal)
533
+ : await runIndividually(adapter, items, flags, model, directive, reporter, budget, fatal);
478
534
  if (fatal.message !== null) {
479
535
  // an environment problem, not a routing result: no summary, no reports
480
536
  io.stderr.write(`skillcheck: stopped, the agent cannot run: ${fatal.message}\n`);
@@ -485,6 +541,7 @@ async function runCommand(file, flags, io, deps) {
485
541
  for (const item of skipped)
486
542
  reporter.caseSkipped(item.c);
487
543
  const notStartedRuns = skipped.reduce((n, item) => n + item.runs.length - item.done, 0);
544
+ const skippedRunCosts = skipped.flatMap((item) => item.runs.flatMap((v) => (v ? [v.costUsd] : [])));
488
545
  const ordered = outcome.results.filter((r) => r !== undefined);
489
546
  const expected = expectedNames(selected);
490
547
  const unavailable = outcome.sawAvailability ? expected.filter((name) => !outcome.available.has(name)) : [];
@@ -495,6 +552,7 @@ async function runCommand(file, flags, io, deps) {
495
552
  skipped: skipped.length,
496
553
  budget: skipped.length > 0 ? { limitUsd: flags.budget, spent: budget.spent, notStartedRuns } : undefined,
497
554
  estimatedUsd: budget.spent,
555
+ skippedRunCosts,
498
556
  });
499
557
  if (flags.batch) {
500
558
  // cases that only errored have no answer to confirm
@@ -513,7 +571,9 @@ async function runCommand(file, flags, io, deps) {
513
571
  cases: items.map((item) => ({ c: item.c, threshold: item.threshold, result: outcome.results[item.jobIndex] ?? null })),
514
572
  unavailable,
515
573
  confusion: pairs,
574
+ batch: flags.batch,
516
575
  estimatedCostUsd: budget.spent,
576
+ skippedRunCosts,
517
577
  budgetUsd: flags.budget ?? null,
518
578
  budgetReached: skipped.length > 0,
519
579
  });
@@ -526,6 +586,8 @@ async function runCommand(file, flags, io, deps) {
526
586
  }
527
587
  if (flags.junit !== undefined)
528
588
  writeReportFile(flags.junit, toJunit(report), "--junit");
589
+ if (flags.markdown !== undefined)
590
+ writeReportFile(flags.markdown, toMarkdown(report), "--markdown");
529
591
  return ordered.some((r) => !r.ok) || skipped.length > 0 ? 1 : 0;
530
592
  }
531
593
  function writeReportFile(path, text, flag) {
@@ -537,9 +599,9 @@ function writeReportFile(path, text, flag) {
537
599
  }
538
600
  }
539
601
  /** One agent call per case: the normal mode. */
540
- async function runIndividually(adapter, items, flags, model, directive, repeat, reporter, budget, fatal) {
602
+ async function runIndividually(adapter, items, flags, model, directive, reporter, budget, fatal) {
541
603
  const totalRuns = items.reduce((n, item) => n + item.runs.length, 0);
542
- reporter.header(items.length, repeat, 1, totalRuns);
604
+ reporter.header(items.length, repeatLabel(items), 1, totalRuns);
543
605
  const jobs = items.flatMap((item, jobIndex) => item.runs.map((_, runIndex) => ({ item, runIndex, jobIndex })));
544
606
  // keep cases in file order even though runs of different cases interleave
545
607
  const results = new Array(items.length);
@@ -575,17 +637,19 @@ async function runIndividually(adapter, items, flags, model, directive, repeat,
575
637
  }
576
638
  /** One agent call per chunk of cases; the model states its choice per request
577
639
  * instead of loading skills. Verdicts flow through the same judge. */
578
- async function runBatched(runBatch, items, flags, model, repeat, reporter, budget, fatal) {
640
+ async function runBatched(runBatch, items, flags, model, reporter, budget, fatal) {
579
641
  const chunks = [];
580
642
  for (let i = 0; i < items.length; i += flags.batchSize)
581
643
  chunks.push(items.slice(i, i + flags.batchSize));
582
644
  const rounds = Math.max(...items.map((item) => item.runs.length));
583
645
  const jobs = [];
584
646
  for (let round = 0; round < rounds; round++) {
647
+ // a round past every repeat in the chunk makes no call
585
648
  for (const chunk of chunks)
586
- jobs.push({ chunk, round });
649
+ if (chunk.some((item) => round < item.runs.length))
650
+ jobs.push({ chunk, round });
587
651
  }
588
- reporter.batchHeader(items.length, repeat, chunks.length * rounds);
652
+ reporter.batchHeader(items.length, repeatLabel(items), jobs.length);
589
653
  const results = new Array(items.length);
590
654
  await runPool(jobs, flags.jobs, async (job) => {
591
655
  // a case with a smaller repeat only takes its first rounds
@@ -597,6 +661,8 @@ async function runBatched(runBatch, items, flags, model, repeat, reporter, budge
597
661
  const parsed = parseBatchAnswer(br.structured, br.text);
598
662
  const chunkError = br.error ?? parsed.error; // a chunk error is every case's error
599
663
  fatal.note(br.error);
664
+ // one call, one budget entry: its usage prices it when the cost never came
665
+ budget.add(br.costUsd, br.usage ?? null);
600
666
  active.forEach((item, i) => {
601
667
  const answer = parsed.answers.get(i + 1);
602
668
  const r = {
@@ -609,7 +675,6 @@ async function runBatched(runBatch, items, flags, model, repeat, reporter, budge
609
675
  durationMs: br.durationMs,
610
676
  };
611
677
  const verdict = judge(item.c, r);
612
- budget.add(verdict.costUsd);
613
678
  item.runs[job.round] = verdict;
614
679
  item.done++;
615
680
  if (item.done === item.runs.length) {
@@ -621,6 +686,13 @@ async function runBatched(runBatch, items, flags, model, repeat, reporter, budge
621
686
  }, { shouldStart: () => fatal.canStart(budget) });
622
687
  return { results, available: new Set(), sawAvailability: false };
623
688
  }
689
+ /** "3", or "1–3" when cases override the repeat. */
690
+ function repeatLabel(items) {
691
+ const counts = items.map((item) => item.runs.length);
692
+ const min = Math.min(...counts);
693
+ const max = Math.max(...counts);
694
+ return min === max ? String(min) : `${min}–${max}`;
695
+ }
624
696
  function selectCases(suite, agent, only) {
625
697
  const forAgent = suite.cases.filter((c) => !c.agents || c.agents.includes(agent));
626
698
  if (!only)
package/dist/import.js ADDED
@@ -0,0 +1,50 @@
1
+ import * as path from "node:path";
2
+ import { parse as parseYaml } from "yaml";
3
+ import { ConfigError, parseSuite } from "./cases.js";
4
+ const PLAIN_NAME = /^[A-Za-z0-9._:/-]+$/;
5
+ /**
6
+ * A skill-creator trigger eval set ([{query, should_trigger}], one skill) as a
7
+ * cases file: should_trigger true becomes expect, false becomes forbid.
8
+ */
9
+ export function evalSetToSuite(data, skill, source) {
10
+ const errors = [];
11
+ if (!Array.isArray(data) || data.length === 0) {
12
+ throw new ConfigError(["expected a JSON array of {query, should_trigger}"]);
13
+ }
14
+ const items = [];
15
+ data.forEach((item, i) => {
16
+ const n = i + 1;
17
+ if (typeof item !== "object" || item === null || Array.isArray(item)) {
18
+ errors.push(`item ${n}: expected an object with query and should_trigger`);
19
+ return;
20
+ }
21
+ const { query, should_trigger: trigger } = item;
22
+ if (typeof query !== "string" || query.trim() === "")
23
+ errors.push(`item ${n}: query must be a non-empty string`);
24
+ if (typeof trigger !== "boolean")
25
+ errors.push(`item ${n}: should_trigger must be true or false`);
26
+ if (typeof query === "string" && typeof trigger === "boolean")
27
+ items.push({ query, trigger });
28
+ });
29
+ if (errors.length > 0)
30
+ throw new ConfigError(errors);
31
+ // plain only when YAML reads it back as the same string: not true, null, 123
32
+ const name = PLAIN_NAME.test(skill) && parseYaml(skill) === skill ? skill : JSON.stringify(skill);
33
+ const lines = [
34
+ `# skillcheck cases imported from ${path.basename(source)} (skill-creator trigger eval set).`,
35
+ "# should_trigger: true became expect, false became forbid: another skill may",
36
+ `# still load there, only ${skill} must not.`,
37
+ "agent: claude",
38
+ "repeat: 1",
39
+ "threshold: 1.0",
40
+ "cases:",
41
+ ];
42
+ for (const { query, trigger } of items) {
43
+ // a JSON string is a valid YAML double-quoted scalar, newlines and quotes included
44
+ lines.push(` - query: ${JSON.stringify(query)}`, ` ${trigger ? "expect" : "forbid"}: [${name}]`);
45
+ }
46
+ const text = lines.join("\n") + "\n";
47
+ parseSuite(parseYaml(text));
48
+ const positive = items.filter((i) => i.trigger).length;
49
+ return { text, positive, negative: items.length - positive };
50
+ }
@@ -0,0 +1,135 @@
1
+ /** An HTML comment so GitHub renders the report body clean under it. */
2
+ export const MARKDOWN_MARKER = "<!-- skillcheck -->";
3
+ /** Failure-focused Markdown summary for PR comments and GitHub job summaries. */
4
+ export function toMarkdown(report) {
5
+ const s = report.summary;
6
+ const lines = [MARKDOWN_MARKER, header(report)];
7
+ lines.push("");
8
+ lines.push(metaLine(report));
9
+ if (report.batch) {
10
+ lines.push("");
11
+ lines.push("> Batch mode: answers are the model's stated choice, not an actual Skill call. Confirm failures with a normal run.");
12
+ }
13
+ if (s.budgetReached && s.budgetUsd !== null) {
14
+ lines.push("");
15
+ lines.push(`> ⚠️ Budget $${s.budgetUsd.toFixed(2)} reached, remaining cases were skipped.`);
16
+ }
17
+ if (report.unavailable.length > 0) {
18
+ lines.push("");
19
+ lines.push(`> ⚠️ Expected skills not available to the agent: ${report.unavailable.join(", ")}`);
20
+ }
21
+ if (s.failed + s.skipped > 0) {
22
+ lines.push("");
23
+ lines.push("| | Case | Request | Loaded | Reason |");
24
+ lines.push("|---|---|---|---|---|");
25
+ for (const c of report.cases) {
26
+ if (c.status === "passed")
27
+ continue;
28
+ lines.push(failedRow(c));
29
+ }
30
+ }
31
+ const passed = report.cases.filter((c) => c.status === "passed");
32
+ if (passed.length > 0) {
33
+ lines.push("");
34
+ lines.push(`<details><summary>Passed (${passed.length})</summary>`);
35
+ lines.push("");
36
+ lines.push("| Case | Request | Loaded |");
37
+ lines.push("|---|---|---|");
38
+ for (const c of passed)
39
+ lines.push(passedRow(c));
40
+ lines.push("");
41
+ lines.push("</details>");
42
+ }
43
+ if (report.confusion.length > 0) {
44
+ lines.push("");
45
+ lines.push("**Confusion** — descriptions to rewrite:");
46
+ lines.push("");
47
+ for (const p of report.confusion)
48
+ lines.push(`- expected \`${p.expected}\` → got \`${p.got}\` (${p.count})`);
49
+ }
50
+ return lines.join("\n") + "\n";
51
+ }
52
+ /** ✅ with the pass count alone, ❌ with failed and skipped counts. */
53
+ function header(report) {
54
+ const s = report.summary;
55
+ if (s.failed === 0 && s.skipped === 0) {
56
+ return `### ✅ skillcheck: ${s.cases} passed`;
57
+ }
58
+ const failedPart = s.failed > 0 ? `${s.failed} failed` : null;
59
+ const skippedPart = s.skipped > 0 ? `${s.skipped} skipped` : null;
60
+ const parts = [failedPart, skippedPart].filter((p) => p !== null).join(", ");
61
+ return `### ❌ skillcheck: ${parts} of ${s.cases}`;
62
+ }
63
+ /** Model, batch flag, runs, cost and duration, " · " separated. */
64
+ function metaLine(report) {
65
+ const s = report.summary;
66
+ const parts = [];
67
+ if (report.model !== null)
68
+ parts.push(report.model);
69
+ if (report.batch)
70
+ parts.push("batch mode");
71
+ parts.push(`${s.runs} run${s.runs === 1 ? "" : "s"}`);
72
+ if (s.unknownCostRuns > 0 && s.estimatedCostUsd > s.costUsd)
73
+ parts.push(`cost ~$${s.estimatedCostUsd.toFixed(2)}`);
74
+ else
75
+ parts.push(`cost $${s.costUsd.toFixed(2)}`);
76
+ parts.push(duration(report.durationMs));
77
+ return parts.join(" · ");
78
+ }
79
+ /** Below a minute in seconds, then minutes and seconds, seconds rounded down. */
80
+ function duration(ms) {
81
+ const total = Math.floor(ms / 1000);
82
+ if (total < 60)
83
+ return `${total}s`;
84
+ return `${Math.floor(total / 60)}m ${total % 60}s`;
85
+ }
86
+ /** ⏭️ for skipped, ⚠️ when every failed run errored, ❌ for the rest. */
87
+ function failedRow(c) {
88
+ const cells = c.status === "skipped"
89
+ ? ["⏭️", caseLabel(c), oneLine(c.query), loaded(c), "budget reached"]
90
+ : [failMark(c), caseLabel(c), oneLine(c.query), loaded(c), reasonCell(c)];
91
+ return `| ${cells.map(esc).join(" | ")} |`;
92
+ }
93
+ /** ⚠️ when every non-ok run errored (the model never answered), ❌ otherwise. */
94
+ function failMark(c) {
95
+ const bad = c.runs.filter((r) => !r.ok);
96
+ return bad.length > 0 && bad.every((r) => r.error !== null) ? "⚠️" : "❌";
97
+ }
98
+ /** The first non-ok run's reason, the pass share prefixed over several runs,
99
+ * a diagnosis appended in italics, then the case note. */
100
+ function reasonCell(c) {
101
+ const bad = c.runs.filter((r) => !r.ok);
102
+ const parts = c.runs.length > 1 ? [`${c.passed}/${c.runs.length}`, bad[0]?.reason ?? ""] : [bad[0]?.reason ?? ""];
103
+ let text = parts.join(" · ");
104
+ const diagnosis = bad.find((r) => r.diagnosis !== null)?.diagnosis;
105
+ if (diagnosis)
106
+ text += ` — _${diagnosis}_`;
107
+ if (c.note !== null)
108
+ text += ` · note: ${c.note}`;
109
+ return text;
110
+ }
111
+ function passedRow(c) {
112
+ const score = c.runs.length > 1 ? ` (${c.passed}/${c.runs.length})` : "";
113
+ const cells = [`${caseLabel(c)}${score}`, oneLine(c.query), loaded(c)];
114
+ return `| ${cells.map(esc).join(" | ")} |`;
115
+ }
116
+ function caseLabel(c) {
117
+ return `#${c.id ?? c.index}`;
118
+ }
119
+ /** Unique loaded skills across runs, "—" when none. */
120
+ function loaded(c) {
121
+ const names = [...new Set(c.runs.flatMap((r) => r.loaded))];
122
+ return names.length > 0 ? names.join(", ") : "—";
123
+ }
124
+ /** Collapse to one line, truncated to 80 chars with "…". */
125
+ function oneLine(text) {
126
+ const flat = text.replace(/\s+/g, " ").trim();
127
+ return flat.length > 80 ? `${flat.slice(0, 80)}…` : flat;
128
+ }
129
+ /** Pipes so they cannot break the table, newlines as spaces, HTML inert. */
130
+ function esc(text) {
131
+ return text
132
+ .replace(/\|/g, "\\|")
133
+ .replace(/[\r\n]+/g, " ")
134
+ .replace(/</g, "&lt;");
135
+ }
package/dist/report.js CHANGED
@@ -11,7 +11,7 @@ export class Reporter {
11
11
  this.useColor = out.isTTY === true && !process.env.NO_COLOR;
12
12
  }
13
13
  header(nCases, repeat, nAgents, totalRuns) {
14
- const runs = totalRuns ?? nCases * repeat * nAgents;
14
+ const runs = totalRuns ?? nCases * Number(repeat) * nAgents;
15
15
  const agents = nAgents === 1 ? "agent" : "agents";
16
16
  this.write(`${nCases} cases × ${repeat} repeat × ${nAgents} ${agents} = ${runs} runs\n`);
17
17
  }
@@ -36,6 +36,8 @@ export class Reporter {
36
36
  if (diagnosed?.diagnosis) {
37
37
  this.write(` ${this.paint("33", `diagnosis ${label}: ${diagnosed.diagnosis}`)}\n`);
38
38
  }
39
+ if (!res.ok && res.case.note)
40
+ this.write(` ${this.paint("2", `note ${label}: ${oneLine(res.case.note)}`)}\n`);
39
41
  }
40
42
  /** A case the budget never let start; printed after all real runs settle. */
41
43
  caseSkipped(c) {
@@ -68,7 +70,8 @@ export class Reporter {
68
70
  const failed = results.filter((r) => !r.ok).length;
69
71
  const skipped = extra?.skipped ?? 0;
70
72
  const skippedPart = skipped > 0 ? `, ${skipped} skipped` : "";
71
- this.write(`${failed} failed${skippedPart} of ${results.length + skipped} · runs ${verdicts.length} · ${costLine(verdicts.map((v) => v.costUsd), extra?.estimatedUsd)}\n`);
73
+ const costs = [...verdicts.map((v) => v.costUsd), ...(extra?.skippedRunCosts ?? [])];
74
+ this.write(`${failed} failed${skippedPart} of ${results.length + skipped} · runs ${costs.length} · ${costLine(costs, extra?.estimatedUsd)}\n`);
72
75
  }
73
76
  paint(code, text) {
74
77
  return this.useColor ? `\u001b[${code}m${text}${RESET}` : text;
package/dist/results.js CHANGED
@@ -2,7 +2,7 @@ import { readVersion } from "./version.js";
2
2
  /** The single source of truth for --json and --junit; computed once per run. */
3
3
  export function buildReport(input) {
4
4
  const verdicts = input.cases.flatMap((e) => e.result?.runs ?? []);
5
- const costs = verdicts.map((v) => v.costUsd);
5
+ const costs = [...verdicts.map((v) => v.costUsd), ...(input.skippedRunCosts ?? [])];
6
6
  const known = costs.filter((c) => c !== null);
7
7
  return {
8
8
  tool: "skillcheck",
@@ -10,13 +10,14 @@ export function buildReport(input) {
10
10
  file: input.file,
11
11
  agent: input.agent,
12
12
  model: input.model,
13
+ batch: input.batch ?? false,
13
14
  startedAt: new Date(input.startedAtMs).toISOString(),
14
15
  durationMs: input.durationMs,
15
16
  summary: {
16
17
  cases: input.cases.length,
17
18
  failed: input.cases.filter((e) => e.result && !e.result.ok).length,
18
19
  skipped: input.cases.filter((e) => !e.result).length,
19
- runs: verdicts.length,
20
+ runs: costs.length,
20
21
  costUsd: known.reduce((a, b) => a + b, 0),
21
22
  unknownCostRuns: costs.length - known.length,
22
23
  estimatedCostUsd: input.estimatedCostUsd,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@icntswm/skillcheck",
3
- "version": "1.0.4",
3
+ "version": "1.2.0",
4
4
  "description": "Regression tests for Claude Code skills: check that every request still loads the right skill",
5
5
  "type": "module",
6
6
  "bin": {