@icntswm/skillcheck 1.0.3 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -4
- package/dist/budget.js +3 -2
- package/dist/cli.js +7 -0
- package/dist/junit.js +2 -1
- package/dist/markdown.js +133 -0
- package/dist/results.js +1 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -36,9 +36,10 @@ requests.
|
|
|
36
36
|
from a broken one instead of letting you guess.
|
|
37
37
|
- 🏷️ **Catches renames and typos.** Case names are checked against the skills
|
|
38
38
|
the agent really has, so a renamed skill can't pass silently.
|
|
39
|
-
- ⚙️ **CI-ready.** A GitHub Action (`uses: icntswm/skillcheck@v1`)
|
|
40
|
-
|
|
41
|
-
`--config-dir` to test
|
|
39
|
+
- ⚙️ **CI-ready.** A GitHub Action (`uses: icntswm/skillcheck@v1`) that
|
|
40
|
+
comments the report on the pull request, JUnit, JSON and Markdown reports,
|
|
41
|
+
clear exit codes, a spending cap (`--budget`), and `--config-dir` to test
|
|
42
|
+
only the skills in your repository.
|
|
42
43
|
- 📄 **Plain YAML, one dependency.** Cases are readable by anyone on the team
|
|
43
44
|
and live next to the skills they test.
|
|
44
45
|
|
|
@@ -125,7 +126,7 @@ stops Claude Code before the first model call, so it costs nothing.
|
|
|
125
126
|
|---|---|
|
|
126
127
|
| [Writing cases](docs/writing-cases.md) | the file format and what makes a case catch regressions |
|
|
127
128
|
| [Cost](docs/cost.md) | what a run costs and how to spend less |
|
|
128
|
-
| [Reports and CI](docs/ci.md) | confusion block, JSON, JUnit, GitHub Actions, exit codes |
|
|
129
|
+
| [Reports and CI](docs/ci.md) | confusion block, JSON, JUnit, Markdown, GitHub Actions, exit codes |
|
|
129
130
|
| [How it works](docs/how-it-works.md) | what happens inside a run, and the limits of each level |
|
|
130
131
|
|
|
131
132
|
`skillcheck --help` lists every command and option.
|
package/dist/budget.js
CHANGED
|
@@ -18,8 +18,9 @@ function listInputPrice(model) {
|
|
|
18
18
|
return 5e-6;
|
|
19
19
|
}
|
|
20
20
|
/**
|
|
21
|
-
* Running spend estimate for --budget.
|
|
22
|
-
*
|
|
21
|
+
* Running spend estimate for --budget. claude usually reports its cost even
|
|
22
|
+
* when early-stop kills it, but a run killed before the result event does
|
|
23
|
+
* not; such runs are priced from their token usage: at the rate seen in
|
|
23
24
|
* finished runs, or at list price before any run finishes. Runs with neither
|
|
24
25
|
* cost nor usage count at the average known cost.
|
|
25
26
|
*/
|
package/dist/cli.js
CHANGED
|
@@ -13,6 +13,7 @@ import { loadSkillDocs } from "./describe.js";
|
|
|
13
13
|
import { aggregate, judge } from "./judge.js";
|
|
14
14
|
import { toJunit } from "./junit.js";
|
|
15
15
|
import { lint, LINT_DEFAULTS } from "./lint.js";
|
|
16
|
+
import { toMarkdown } from "./markdown.js";
|
|
16
17
|
import { runPool } from "./pool.js";
|
|
17
18
|
import { oneLine, Reporter } from "./report.js";
|
|
18
19
|
import { buildReport } from "./results.js";
|
|
@@ -42,6 +43,7 @@ run options:
|
|
|
42
43
|
--json <path> machine-readable report to path ("-" writes it to stdout
|
|
43
44
|
and the terminal report to stderr)
|
|
44
45
|
--junit <path> JUnit XML report (GitLab/GitHub test reporters) to path
|
|
46
|
+
--markdown <path> Markdown summary (PR comments, GitHub job summary) to path
|
|
45
47
|
run/check options:
|
|
46
48
|
--skill <a,b> keep only cases that mention these skills
|
|
47
49
|
check options:
|
|
@@ -86,6 +88,7 @@ const OPTIONS = {
|
|
|
86
88
|
budget: { type: "string" },
|
|
87
89
|
json: { type: "string" },
|
|
88
90
|
junit: { type: "string" },
|
|
91
|
+
markdown: { type: "string" },
|
|
89
92
|
top: { type: "string" },
|
|
90
93
|
overlap: { type: "string" },
|
|
91
94
|
strict: { type: "boolean", default: false },
|
|
@@ -120,6 +123,7 @@ function parseFlags(values, cwd) {
|
|
|
120
123
|
budget: numberFlag(values.budget),
|
|
121
124
|
json: values.json,
|
|
122
125
|
junit: values.junit,
|
|
126
|
+
markdown: values.markdown,
|
|
123
127
|
top: intFlag(values.top, "top", 1) ?? 5,
|
|
124
128
|
overlap: ratioFlag(values.overlap) ?? 0.3,
|
|
125
129
|
strict: values.strict,
|
|
@@ -513,6 +517,7 @@ async function runCommand(file, flags, io, deps) {
|
|
|
513
517
|
cases: items.map((item) => ({ c: item.c, threshold: item.threshold, result: outcome.results[item.jobIndex] ?? null })),
|
|
514
518
|
unavailable,
|
|
515
519
|
confusion: pairs,
|
|
520
|
+
batch: flags.batch,
|
|
516
521
|
estimatedCostUsd: budget.spent,
|
|
517
522
|
budgetUsd: flags.budget ?? null,
|
|
518
523
|
budgetReached: skipped.length > 0,
|
|
@@ -526,6 +531,8 @@ async function runCommand(file, flags, io, deps) {
|
|
|
526
531
|
}
|
|
527
532
|
if (flags.junit !== undefined)
|
|
528
533
|
writeReportFile(flags.junit, toJunit(report), "--junit");
|
|
534
|
+
if (flags.markdown !== undefined)
|
|
535
|
+
writeReportFile(flags.markdown, toMarkdown(report), "--markdown");
|
|
529
536
|
return ordered.some((r) => !r.ok) || skipped.length > 0 ? 1 : 0;
|
|
530
537
|
}
|
|
531
538
|
function writeReportFile(path, text, flag) {
|
package/dist/junit.js
CHANGED
|
@@ -14,9 +14,10 @@ export function toJunit(report) {
|
|
|
14
14
|
if (report.model !== null)
|
|
15
15
|
properties.push(` <property name="model" value="${esc(report.model)}"/>`);
|
|
16
16
|
properties.push(` <property name="costUsd" value="${report.summary.costUsd.toFixed(2)}"/>`);
|
|
17
|
-
//
|
|
17
|
+
// a run killed before its result event reports no cost, so costUsd alone understates the spend
|
|
18
18
|
if (report.summary.unknownCostRuns > 0) {
|
|
19
19
|
properties.push(` <property name="unknownCostRuns" value="${report.summary.unknownCostRuns}"/>`);
|
|
20
|
+
properties.push(` <property name="estimatedCostUsd" value="${report.summary.estimatedCostUsd.toFixed(2)}"/>`);
|
|
20
21
|
}
|
|
21
22
|
properties.push(" </properties>");
|
|
22
23
|
const cases = rows.map(({ c, kind }) => testcase(c, kind, suiteName));
|
package/dist/markdown.js
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
/** An HTML comment so GitHub renders the report body clean under it. */
|
|
2
|
+
export const MARKDOWN_MARKER = "<!-- skillcheck -->";
|
|
3
|
+
/** Failure-focused Markdown summary for PR comments and GitHub job summaries. */
|
|
4
|
+
export function toMarkdown(report) {
|
|
5
|
+
const s = report.summary;
|
|
6
|
+
const lines = [MARKDOWN_MARKER, header(report)];
|
|
7
|
+
lines.push("");
|
|
8
|
+
lines.push(metaLine(report));
|
|
9
|
+
if (report.batch) {
|
|
10
|
+
lines.push("");
|
|
11
|
+
lines.push("> Batch mode: answers are the model's stated choice, not an actual Skill call. Confirm failures with a normal run.");
|
|
12
|
+
}
|
|
13
|
+
if (s.budgetReached && s.budgetUsd !== null) {
|
|
14
|
+
lines.push("");
|
|
15
|
+
lines.push(`> ⚠️ Budget $${s.budgetUsd.toFixed(2)} reached, remaining cases were skipped.`);
|
|
16
|
+
}
|
|
17
|
+
if (report.unavailable.length > 0) {
|
|
18
|
+
lines.push("");
|
|
19
|
+
lines.push(`> ⚠️ Expected skills not available to the agent: ${report.unavailable.join(", ")}`);
|
|
20
|
+
}
|
|
21
|
+
if (s.failed + s.skipped > 0) {
|
|
22
|
+
lines.push("");
|
|
23
|
+
lines.push("| | Case | Request | Loaded | Reason |");
|
|
24
|
+
lines.push("|---|---|---|---|---|");
|
|
25
|
+
for (const c of report.cases) {
|
|
26
|
+
if (c.status === "passed")
|
|
27
|
+
continue;
|
|
28
|
+
lines.push(failedRow(c));
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
const passed = report.cases.filter((c) => c.status === "passed");
|
|
32
|
+
if (passed.length > 0) {
|
|
33
|
+
lines.push("");
|
|
34
|
+
lines.push(`<details><summary>Passed (${passed.length})</summary>`);
|
|
35
|
+
lines.push("");
|
|
36
|
+
lines.push("| Case | Request | Loaded |");
|
|
37
|
+
lines.push("|---|---|---|");
|
|
38
|
+
for (const c of passed)
|
|
39
|
+
lines.push(passedRow(c));
|
|
40
|
+
lines.push("");
|
|
41
|
+
lines.push("</details>");
|
|
42
|
+
}
|
|
43
|
+
if (report.confusion.length > 0) {
|
|
44
|
+
lines.push("");
|
|
45
|
+
lines.push("**Confusion** — descriptions to rewrite:");
|
|
46
|
+
lines.push("");
|
|
47
|
+
for (const p of report.confusion)
|
|
48
|
+
lines.push(`- expected \`${p.expected}\` → got \`${p.got}\` (${p.count})`);
|
|
49
|
+
}
|
|
50
|
+
return lines.join("\n") + "\n";
|
|
51
|
+
}
|
|
52
|
+
/** ✅ with the pass count alone, ❌ with failed and skipped counts. */
|
|
53
|
+
function header(report) {
|
|
54
|
+
const s = report.summary;
|
|
55
|
+
if (s.failed === 0 && s.skipped === 0) {
|
|
56
|
+
return `### ✅ skillcheck: ${s.cases} passed`;
|
|
57
|
+
}
|
|
58
|
+
const failedPart = s.failed > 0 ? `${s.failed} failed` : null;
|
|
59
|
+
const skippedPart = s.skipped > 0 ? `${s.skipped} skipped` : null;
|
|
60
|
+
const parts = [failedPart, skippedPart].filter((p) => p !== null).join(", ");
|
|
61
|
+
return `### ❌ skillcheck: ${parts} of ${s.cases}`;
|
|
62
|
+
}
|
|
63
|
+
/** Model, batch flag, runs, cost and duration, " · " separated. */
|
|
64
|
+
function metaLine(report) {
|
|
65
|
+
const s = report.summary;
|
|
66
|
+
const parts = [];
|
|
67
|
+
if (report.model !== null)
|
|
68
|
+
parts.push(report.model);
|
|
69
|
+
if (report.batch)
|
|
70
|
+
parts.push("batch mode");
|
|
71
|
+
parts.push(`${s.runs} run${s.runs === 1 ? "" : "s"}`);
|
|
72
|
+
if (s.unknownCostRuns > 0 && s.estimatedCostUsd > s.costUsd)
|
|
73
|
+
parts.push(`cost ~$${s.estimatedCostUsd.toFixed(2)}`);
|
|
74
|
+
else
|
|
75
|
+
parts.push(`cost $${s.costUsd.toFixed(2)}`);
|
|
76
|
+
parts.push(duration(report.durationMs));
|
|
77
|
+
return parts.join(" · ");
|
|
78
|
+
}
|
|
79
|
+
/** Below a minute in seconds, then minutes and seconds, seconds rounded down. */
|
|
80
|
+
function duration(ms) {
|
|
81
|
+
const total = Math.floor(ms / 1000);
|
|
82
|
+
if (total < 60)
|
|
83
|
+
return `${total}s`;
|
|
84
|
+
return `${Math.floor(total / 60)}m ${total % 60}s`;
|
|
85
|
+
}
|
|
86
|
+
/** ⏭️ for skipped, ⚠️ when every failed run errored, ❌ for the rest. */
|
|
87
|
+
function failedRow(c) {
|
|
88
|
+
const cells = c.status === "skipped"
|
|
89
|
+
? ["⏭️", caseLabel(c), oneLine(c.query), loaded(c), "budget reached"]
|
|
90
|
+
: [failMark(c), caseLabel(c), oneLine(c.query), loaded(c), reasonCell(c)];
|
|
91
|
+
return `| ${cells.map(esc).join(" | ")} |`;
|
|
92
|
+
}
|
|
93
|
+
/** ⚠️ when every non-ok run errored (the model never answered), ❌ otherwise. */
|
|
94
|
+
function failMark(c) {
|
|
95
|
+
const bad = c.runs.filter((r) => !r.ok);
|
|
96
|
+
return bad.length > 0 && bad.every((r) => r.error !== null) ? "⚠️" : "❌";
|
|
97
|
+
}
|
|
98
|
+
/** The first non-ok run's reason, the pass share prefixed over several runs,
|
|
99
|
+
* a diagnosis appended in italics. */
|
|
100
|
+
function reasonCell(c) {
|
|
101
|
+
const bad = c.runs.filter((r) => !r.ok);
|
|
102
|
+
const parts = c.runs.length > 1 ? [`${c.passed}/${c.runs.length}`, bad[0]?.reason ?? ""] : [bad[0]?.reason ?? ""];
|
|
103
|
+
let text = parts.join(" · ");
|
|
104
|
+
const diagnosis = bad.find((r) => r.diagnosis !== null)?.diagnosis;
|
|
105
|
+
if (diagnosis)
|
|
106
|
+
text += ` — _${diagnosis}_`;
|
|
107
|
+
return text;
|
|
108
|
+
}
|
|
109
|
+
function passedRow(c) {
|
|
110
|
+
const score = c.runs.length > 1 ? ` (${c.passed}/${c.runs.length})` : "";
|
|
111
|
+
const cells = [`${caseLabel(c)}${score}`, oneLine(c.query), loaded(c)];
|
|
112
|
+
return `| ${cells.map(esc).join(" | ")} |`;
|
|
113
|
+
}
|
|
114
|
+
function caseLabel(c) {
|
|
115
|
+
return `#${c.id ?? c.index}`;
|
|
116
|
+
}
|
|
117
|
+
/** Unique loaded skills across runs, "—" when none. */
|
|
118
|
+
function loaded(c) {
|
|
119
|
+
const names = [...new Set(c.runs.flatMap((r) => r.loaded))];
|
|
120
|
+
return names.length > 0 ? names.join(", ") : "—";
|
|
121
|
+
}
|
|
122
|
+
/** Collapse to one line, truncated to 80 chars with "…". */
|
|
123
|
+
function oneLine(text) {
|
|
124
|
+
const flat = text.replace(/\s+/g, " ").trim();
|
|
125
|
+
return flat.length > 80 ? `${flat.slice(0, 80)}…` : flat;
|
|
126
|
+
}
|
|
127
|
+
/** Pipes so they cannot break the table, newlines as spaces, HTML inert. */
|
|
128
|
+
function esc(text) {
|
|
129
|
+
return text
|
|
130
|
+
.replace(/\|/g, "\\|")
|
|
131
|
+
.replace(/[\r\n]+/g, " ")
|
|
132
|
+
.replace(/</g, "<");
|
|
133
|
+
}
|
package/dist/results.js
CHANGED