@icntswm/skillcheck 1.0.4 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -36,9 +36,10 @@ requests.
36
36
  from a broken one instead of letting you guess.
37
37
  - 🏷️ **Catches renames and typos.** Case names are checked against the skills
38
38
  the agent really has, so a renamed skill can't pass silently.
39
- - ⚙️ **CI-ready.** A GitHub Action (`uses: icntswm/skillcheck@v1`), JUnit
40
- and JSON reports, clear exit codes, a spending cap (`--budget`), and
41
- `--config-dir` to test only the skills in your repository.
39
+ - ⚙️ **CI-ready.** A GitHub Action (`uses: icntswm/skillcheck@v1`) that
40
+ comments the report on the pull request, JUnit, JSON and Markdown reports,
41
+ clear exit codes, a spending cap (`--budget`), and `--config-dir` to test
42
+ only the skills in your repository.
42
43
  - 📄 **Plain YAML, one dependency.** Cases are readable by anyone on the team
43
44
  and live next to the skills they test.
44
45
 
@@ -125,7 +126,7 @@ stops Claude Code before the first model call, so it costs nothing.
125
126
  |---|---|
126
127
  | [Writing cases](docs/writing-cases.md) | the file format and what makes a case catch regressions |
127
128
  | [Cost](docs/cost.md) | what a run costs and how to spend less |
128
- | [Reports and CI](docs/ci.md) | confusion block, JSON, JUnit, GitHub Actions, exit codes |
129
+ | [Reports and CI](docs/ci.md) | confusion block, JSON, JUnit, Markdown, GitHub Actions, exit codes |
129
130
  | [How it works](docs/how-it-works.md) | what happens inside a run, and the limits of each level |
130
131
 
131
132
  `skillcheck --help` lists every command and option.
package/dist/cli.js CHANGED
@@ -13,6 +13,7 @@ import { loadSkillDocs } from "./describe.js";
13
13
  import { aggregate, judge } from "./judge.js";
14
14
  import { toJunit } from "./junit.js";
15
15
  import { lint, LINT_DEFAULTS } from "./lint.js";
16
+ import { toMarkdown } from "./markdown.js";
16
17
  import { runPool } from "./pool.js";
17
18
  import { oneLine, Reporter } from "./report.js";
18
19
  import { buildReport } from "./results.js";
@@ -42,6 +43,7 @@ run options:
42
43
  --json <path> machine-readable report to path ("-" writes it to stdout
43
44
  and the terminal report to stderr)
44
45
  --junit <path> JUnit XML report (GitLab/GitHub test reporters) to path
46
+ --markdown <path> Markdown summary (PR comments, GitHub job summary) to path
45
47
  run/check options:
46
48
  --skill <a,b> keep only cases that mention these skills
47
49
  check options:
@@ -86,6 +88,7 @@ const OPTIONS = {
86
88
  budget: { type: "string" },
87
89
  json: { type: "string" },
88
90
  junit: { type: "string" },
91
+ markdown: { type: "string" },
89
92
  top: { type: "string" },
90
93
  overlap: { type: "string" },
91
94
  strict: { type: "boolean", default: false },
@@ -120,6 +123,7 @@ function parseFlags(values, cwd) {
120
123
  budget: numberFlag(values.budget),
121
124
  json: values.json,
122
125
  junit: values.junit,
126
+ markdown: values.markdown,
123
127
  top: intFlag(values.top, "top", 1) ?? 5,
124
128
  overlap: ratioFlag(values.overlap) ?? 0.3,
125
129
  strict: values.strict,
@@ -513,6 +517,7 @@ async function runCommand(file, flags, io, deps) {
513
517
  cases: items.map((item) => ({ c: item.c, threshold: item.threshold, result: outcome.results[item.jobIndex] ?? null })),
514
518
  unavailable,
515
519
  confusion: pairs,
520
+ batch: flags.batch,
516
521
  estimatedCostUsd: budget.spent,
517
522
  budgetUsd: flags.budget ?? null,
518
523
  budgetReached: skipped.length > 0,
@@ -526,6 +531,8 @@ async function runCommand(file, flags, io, deps) {
526
531
  }
527
532
  if (flags.junit !== undefined)
528
533
  writeReportFile(flags.junit, toJunit(report), "--junit");
534
+ if (flags.markdown !== undefined)
535
+ writeReportFile(flags.markdown, toMarkdown(report), "--markdown");
529
536
  return ordered.some((r) => !r.ok) || skipped.length > 0 ? 1 : 0;
530
537
  }
531
538
  function writeReportFile(path, text, flag) {
@@ -0,0 +1,133 @@
1
+ /** An HTML comment so GitHub renders the report body clean under it. */
2
+ export const MARKDOWN_MARKER = "<!-- skillcheck -->";
3
+ /** Failure-focused Markdown summary for PR comments and GitHub job summaries. */
4
+ export function toMarkdown(report) {
5
+ const s = report.summary;
6
+ const lines = [MARKDOWN_MARKER, header(report)];
7
+ lines.push("");
8
+ lines.push(metaLine(report));
9
+ if (report.batch) {
10
+ lines.push("");
11
+ lines.push("> Batch mode: answers are the model's stated choice, not an actual Skill call. Confirm failures with a normal run.");
12
+ }
13
+ if (s.budgetReached && s.budgetUsd !== null) {
14
+ lines.push("");
15
+ lines.push(`> ⚠️ Budget $${s.budgetUsd.toFixed(2)} reached, remaining cases were skipped.`);
16
+ }
17
+ if (report.unavailable.length > 0) {
18
+ lines.push("");
19
+ lines.push(`> ⚠️ Expected skills not available to the agent: ${report.unavailable.join(", ")}`);
20
+ }
21
+ if (s.failed + s.skipped > 0) {
22
+ lines.push("");
23
+ lines.push("| | Case | Request | Loaded | Reason |");
24
+ lines.push("|---|---|---|---|---|");
25
+ for (const c of report.cases) {
26
+ if (c.status === "passed")
27
+ continue;
28
+ lines.push(failedRow(c));
29
+ }
30
+ }
31
+ const passed = report.cases.filter((c) => c.status === "passed");
32
+ if (passed.length > 0) {
33
+ lines.push("");
34
+ lines.push(`<details><summary>Passed (${passed.length})</summary>`);
35
+ lines.push("");
36
+ lines.push("| Case | Request | Loaded |");
37
+ lines.push("|---|---|---|");
38
+ for (const c of passed)
39
+ lines.push(passedRow(c));
40
+ lines.push("");
41
+ lines.push("</details>");
42
+ }
43
+ if (report.confusion.length > 0) {
44
+ lines.push("");
45
+ lines.push("**Confusion** — descriptions to rewrite:");
46
+ lines.push("");
47
+ for (const p of report.confusion)
48
+ lines.push(`- expected \`${p.expected}\` → got \`${p.got}\` (${p.count})`);
49
+ }
50
+ return lines.join("\n") + "\n";
51
+ }
52
+ /** ✅ with the pass count alone, ❌ with failed and skipped counts. */
53
+ function header(report) {
54
+ const s = report.summary;
55
+ if (s.failed === 0 && s.skipped === 0) {
56
+ return `### ✅ skillcheck: ${s.cases} passed`;
57
+ }
58
+ const failedPart = s.failed > 0 ? `${s.failed} failed` : null;
59
+ const skippedPart = s.skipped > 0 ? `${s.skipped} skipped` : null;
60
+ const parts = [failedPart, skippedPart].filter((p) => p !== null).join(", ");
61
+ return `### ❌ skillcheck: ${parts} of ${s.cases}`;
62
+ }
63
+ /** Model, batch flag, runs, cost and duration, " · " separated. */
64
+ function metaLine(report) {
65
+ const s = report.summary;
66
+ const parts = [];
67
+ if (report.model !== null)
68
+ parts.push(report.model);
69
+ if (report.batch)
70
+ parts.push("batch mode");
71
+ parts.push(`${s.runs} run${s.runs === 1 ? "" : "s"}`);
72
+ if (s.unknownCostRuns > 0 && s.estimatedCostUsd > s.costUsd)
73
+ parts.push(`cost ~$${s.estimatedCostUsd.toFixed(2)}`);
74
+ else
75
+ parts.push(`cost $${s.costUsd.toFixed(2)}`);
76
+ parts.push(duration(report.durationMs));
77
+ return parts.join(" · ");
78
+ }
79
+ /** Below a minute in seconds, then minutes and seconds, seconds rounded down. */
80
+ function duration(ms) {
81
+ const total = Math.floor(ms / 1000);
82
+ if (total < 60)
83
+ return `${total}s`;
84
+ return `${Math.floor(total / 60)}m ${total % 60}s`;
85
+ }
86
+ /** ⏭️ for skipped, ⚠️ when every failed run errored, ❌ for the rest. */
87
+ function failedRow(c) {
88
+ const cells = c.status === "skipped"
89
+ ? ["⏭️", caseLabel(c), oneLine(c.query), loaded(c), "budget reached"]
90
+ : [failMark(c), caseLabel(c), oneLine(c.query), loaded(c), reasonCell(c)];
91
+ return `| ${cells.map(esc).join(" | ")} |`;
92
+ }
93
+ /** ⚠️ when every non-ok run errored (the model never answered), ❌ otherwise. */
94
+ function failMark(c) {
95
+ const bad = c.runs.filter((r) => !r.ok);
96
+ return bad.length > 0 && bad.every((r) => r.error !== null) ? "⚠️" : "❌";
97
+ }
98
+ /** The first non-ok run's reason, the pass share prefixed over several runs,
99
+ * a diagnosis appended in italics. */
100
+ function reasonCell(c) {
101
+ const bad = c.runs.filter((r) => !r.ok);
102
+ const parts = c.runs.length > 1 ? [`${c.passed}/${c.runs.length}`, bad[0]?.reason ?? ""] : [bad[0]?.reason ?? ""];
103
+ let text = parts.join(" · ");
104
+ const diagnosis = bad.find((r) => r.diagnosis !== null)?.diagnosis;
105
+ if (diagnosis)
106
+ text += ` — _${diagnosis}_`;
107
+ return text;
108
+ }
109
+ function passedRow(c) {
110
+ const score = c.runs.length > 1 ? ` (${c.passed}/${c.runs.length})` : "";
111
+ const cells = [`${caseLabel(c)}${score}`, oneLine(c.query), loaded(c)];
112
+ return `| ${cells.map(esc).join(" | ")} |`;
113
+ }
114
+ function caseLabel(c) {
115
+ return `#${c.id ?? c.index}`;
116
+ }
117
+ /** Unique loaded skills across runs, "—" when none. */
118
+ function loaded(c) {
119
+ const names = [...new Set(c.runs.flatMap((r) => r.loaded))];
120
+ return names.length > 0 ? names.join(", ") : "—";
121
+ }
122
+ /** Collapse to one line, truncated to 80 chars with "…". */
123
+ function oneLine(text) {
124
+ const flat = text.replace(/\s+/g, " ").trim();
125
+ return flat.length > 80 ? `${flat.slice(0, 80)}…` : flat;
126
+ }
127
+ /** Pipes so they cannot break the table, newlines as spaces, HTML inert. */
128
+ function esc(text) {
129
+ return text
130
+ .replace(/\|/g, "\\|")
131
+ .replace(/[\r\n]+/g, " ")
132
+ .replace(/</g, "&lt;");
133
+ }
package/dist/results.js CHANGED
@@ -10,6 +10,7 @@ export function buildReport(input) {
10
10
  file: input.file,
11
11
  agent: input.agent,
12
12
  model: input.model,
13
+ batch: input.batch ?? false,
13
14
  startedAt: new Date(input.startedAtMs).toISOString(),
14
15
  durationMs: input.durationMs,
15
16
  summary: {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@icntswm/skillcheck",
3
- "version": "1.0.4",
3
+ "version": "1.1.0",
4
4
  "description": "Regression tests for Claude Code skills: check that every request still loads the right skill",
5
5
  "type": "module",
6
6
  "bin": {