@icntswm/skillcheck 1.0.4 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -4
- package/dist/agents/claude.js +8 -5
- package/dist/cli.js +81 -9
- package/dist/import.js +50 -0
- package/dist/markdown.js +135 -0
- package/dist/report.js +5 -2
- package/dist/results.js +3 -2
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -36,9 +36,10 @@ requests.
|
|
|
36
36
|
from a broken one instead of letting you guess.
|
|
37
37
|
- 🏷️ **Catches renames and typos.** Case names are checked against the skills
|
|
38
38
|
the agent really has, so a renamed skill can't pass silently.
|
|
39
|
-
- ⚙️ **CI-ready.** A GitHub Action (`uses: icntswm/skillcheck@v1`)
|
|
40
|
-
|
|
41
|
-
`--config-dir` to test
|
|
39
|
+
- ⚙️ **CI-ready.** A GitHub Action (`uses: icntswm/skillcheck@v1`) that
|
|
40
|
+
comments the report on the pull request, JUnit, JSON and Markdown reports,
|
|
41
|
+
clear exit codes, a spending cap (`--budget`), and `--config-dir` to test
|
|
42
|
+
only the skills in your repository.
|
|
42
43
|
- 📄 **Plain YAML, one dependency.** Cases are readable by anyone on the team
|
|
43
44
|
and live next to the skills they test.
|
|
44
45
|
|
|
@@ -119,13 +120,17 @@ skillcheck run --only 2,5 # confirm what the batch flagged
|
|
|
119
120
|
`skillcheck list` prints the skills and slash commands Claude Code sees. It
|
|
120
121
|
stops Claude Code before the first model call, so it costs nothing.
|
|
121
122
|
|
|
123
|
+
Already tuned a description with Anthropic's skill-creator? `skillcheck import
|
|
124
|
+
eval_set.json --skill <name>` turns its trigger eval set into cases (see
|
|
125
|
+
[Writing cases](docs/writing-cases.md#from-skill-creator)).
|
|
126
|
+
|
|
122
127
|
## Documentation
|
|
123
128
|
|
|
124
129
|
| | |
|
|
125
130
|
|---|---|
|
|
126
131
|
| [Writing cases](docs/writing-cases.md) | the file format and what makes a case catch regressions |
|
|
127
132
|
| [Cost](docs/cost.md) | what a run costs and how to spend less |
|
|
128
|
-
| [Reports and CI](docs/ci.md) | confusion block, JSON, JUnit, GitHub Actions, exit codes |
|
|
133
|
+
| [Reports and CI](docs/ci.md) | confusion block, JSON, JUnit, Markdown, GitHub Actions, exit codes |
|
|
129
134
|
| [How it works](docs/how-it-works.md) | what happens inside a run, and the limits of each level |
|
|
130
135
|
|
|
131
136
|
`skillcheck --help` lists every command and option.
|
package/dist/agents/claude.js
CHANGED
|
@@ -327,7 +327,8 @@ export class ClaudeAdapter {
|
|
|
327
327
|
},
|
|
328
328
|
});
|
|
329
329
|
const { text, costUsd, structuredOutput } = out.stream.result;
|
|
330
|
-
|
|
330
|
+
const error = withLoginHint(out.error, opts.configDir);
|
|
331
|
+
return { structured: structuredOutput, text, costUsd, usage: out.stream.usage, error, durationMs: out.durationMs };
|
|
331
332
|
}
|
|
332
333
|
finally {
|
|
333
334
|
await rm(workdir, { recursive: true, force: true });
|
|
@@ -408,13 +409,15 @@ export class ClaudeAdapter {
|
|
|
408
409
|
clearTimeout(killTimer);
|
|
409
410
|
if (timeoutTimer)
|
|
410
411
|
clearTimeout(timeoutTimer);
|
|
411
|
-
// A non-zero exit is not a failure by itself: error_max_turns is normal
|
|
412
|
-
//
|
|
412
|
+
// A non-zero exit is not a failure by itself: error_max_turns is normal
|
|
413
|
+
// and ends with a result event. Without one, the run ended before routing
|
|
414
|
+
// was over, unless we stopped it ourselves.
|
|
413
415
|
if (!error)
|
|
414
416
|
error = stream.error;
|
|
415
|
-
if (!error && !
|
|
417
|
+
if (!error && !stopping && !stream.finished) {
|
|
416
418
|
const detail = stderrText.trim().slice(0, 120);
|
|
417
|
-
|
|
419
|
+
const why = stdoutSeen ? `claude exited with code ${code ?? signal} before its result` : `claude exited with code ${code ?? signal}`;
|
|
420
|
+
error = `${why}${detail ? `: ${detail}` : ""}`;
|
|
418
421
|
}
|
|
419
422
|
resolve({ stream, error, stoppedEarly, durationMs: Date.now() - startedAt });
|
|
420
423
|
};
|
package/dist/cli.js
CHANGED
|
@@ -11,8 +11,10 @@ import { BUILTIN_SKILLS, ConfigError, findDefaultCasesFile, knownSkillNames, loa
|
|
|
11
11
|
import { confusion } from "./confusion.js";
|
|
12
12
|
import { loadSkillDocs } from "./describe.js";
|
|
13
13
|
import { aggregate, judge } from "./judge.js";
|
|
14
|
+
import { evalSetToSuite } from "./import.js";
|
|
14
15
|
import { toJunit } from "./junit.js";
|
|
15
16
|
import { lint, LINT_DEFAULTS } from "./lint.js";
|
|
17
|
+
import { toMarkdown } from "./markdown.js";
|
|
16
18
|
import { runPool } from "./pool.js";
|
|
17
19
|
import { oneLine, Reporter } from "./report.js";
|
|
18
20
|
import { buildReport } from "./results.js";
|
|
@@ -25,6 +27,8 @@ Usage:
|
|
|
25
27
|
skillcheck lint [file] [options] static checks on descriptions and cases
|
|
26
28
|
skillcheck list [options] print the skills and commands the agent can load
|
|
27
29
|
skillcheck init [file] [options] write a starter cases file with those names
|
|
30
|
+
skillcheck import <file> --skill <name>
|
|
31
|
+
turn a skill-creator trigger eval set into cases
|
|
28
32
|
|
|
29
33
|
run options:
|
|
30
34
|
-a, --agent <name> agent to route with (default: suite or claude)
|
|
@@ -42,6 +46,7 @@ run options:
|
|
|
42
46
|
--json <path> machine-readable report to path ("-" writes it to stdout
|
|
43
47
|
and the terminal report to stderr)
|
|
44
48
|
--junit <path> JUnit XML report (GitLab/GitHub test reporters) to path
|
|
49
|
+
--markdown <path> Markdown summary (PR comments, GitHub job summary) to path
|
|
45
50
|
run/check options:
|
|
46
51
|
--skill <a,b> keep only cases that mention these skills
|
|
47
52
|
check options:
|
|
@@ -57,6 +62,9 @@ list/init options:
|
|
|
57
62
|
CLAUDE_CONFIG_DIR; run, list and init also set it for the agent
|
|
58
63
|
init options:
|
|
59
64
|
--force overwrite an existing file
|
|
65
|
+
import options:
|
|
66
|
+
--skill <name> the skill the eval set is about
|
|
67
|
+
-o, --out <file> write the cases there instead of stdout (--force overwrites)
|
|
60
68
|
common:
|
|
61
69
|
-h, --help show this help
|
|
62
70
|
--version show version
|
|
@@ -67,6 +75,7 @@ nothing reaches the model and nothing is billed, even when not logged in.
|
|
|
67
75
|
|
|
68
76
|
Exit codes: 0 all passed, 1 some case failed or was skipped, 2 config or environment error.
|
|
69
77
|
`;
|
|
78
|
+
const COMMANDS = ["run", "check", "lint", "list", "init", "import"];
|
|
70
79
|
const OPTIONS = {
|
|
71
80
|
help: { type: "boolean", short: "h", default: false },
|
|
72
81
|
version: { type: "boolean", default: false },
|
|
@@ -86,11 +95,13 @@ const OPTIONS = {
|
|
|
86
95
|
budget: { type: "string" },
|
|
87
96
|
json: { type: "string" },
|
|
88
97
|
junit: { type: "string" },
|
|
98
|
+
markdown: { type: "string" },
|
|
89
99
|
top: { type: "string" },
|
|
90
100
|
overlap: { type: "string" },
|
|
91
101
|
strict: { type: "boolean", default: false },
|
|
92
102
|
"config-dir": { type: "string" },
|
|
93
103
|
force: { type: "boolean", default: false },
|
|
104
|
+
out: { type: "string", short: "o" },
|
|
94
105
|
};
|
|
95
106
|
/** Config/environment problems: message to stderr, exit 2. */
|
|
96
107
|
class UsageError extends Error {
|
|
@@ -120,11 +131,13 @@ function parseFlags(values, cwd) {
|
|
|
120
131
|
budget: numberFlag(values.budget),
|
|
121
132
|
json: values.json,
|
|
122
133
|
junit: values.junit,
|
|
134
|
+
markdown: values.markdown,
|
|
123
135
|
top: intFlag(values.top, "top", 1) ?? 5,
|
|
124
136
|
overlap: ratioFlag(values.overlap) ?? 0.3,
|
|
125
137
|
strict: values.strict,
|
|
126
138
|
configDir: values["config-dir"] !== undefined ? resolveConfigDir(values["config-dir"], cwd) : undefined,
|
|
127
139
|
force: values.force,
|
|
140
|
+
out: values.out,
|
|
128
141
|
};
|
|
129
142
|
}
|
|
130
143
|
/** --config-dir: absolute, must be an existing directory. */
|
|
@@ -199,7 +212,7 @@ export async function main(argv, io = { stdout: process.stdout, stderr: process.
|
|
|
199
212
|
io.stderr.write(USAGE);
|
|
200
213
|
return 2;
|
|
201
214
|
}
|
|
202
|
-
if (command
|
|
215
|
+
if (!COMMANDS.includes(command)) {
|
|
203
216
|
io.stderr.write(`skillcheck: unknown command "${command}"\n\n${USAGE}`);
|
|
204
217
|
return 2;
|
|
205
218
|
}
|
|
@@ -210,6 +223,8 @@ export async function main(argv, io = { stdout: process.stdout, stderr: process.
|
|
|
210
223
|
return await listCommand(flags, io, deps);
|
|
211
224
|
if (command === "init")
|
|
212
225
|
return await initCommand(positionals[1], flags, io, deps);
|
|
226
|
+
if (command === "import")
|
|
227
|
+
return importCommand(positionals[1], flags, io);
|
|
213
228
|
const file = casesFile(positionals[1], io);
|
|
214
229
|
if (command === "run")
|
|
215
230
|
return await runCommand(file, flags, io, deps);
|
|
@@ -372,6 +387,47 @@ async function initCommand(positional, flags, io, deps) {
|
|
|
372
387
|
io.stdout.write(`wrote ${file} (${skills.length} skills listed)\n`);
|
|
373
388
|
return 0;
|
|
374
389
|
}
|
|
390
|
+
function importCommand(positional, flags, io) {
|
|
391
|
+
if (positional === undefined) {
|
|
392
|
+
throw new UsageError("import needs the eval set file: skillcheck import <eval_set.json> --skill <name>");
|
|
393
|
+
}
|
|
394
|
+
const skill = flags.skill?.trim();
|
|
395
|
+
if (!skill || skill.includes(","))
|
|
396
|
+
throw new UsageError("import needs --skill <name>: the skill the eval set is about");
|
|
397
|
+
const source = path.resolve(io.cwd, positional);
|
|
398
|
+
let data;
|
|
399
|
+
try {
|
|
400
|
+
data = JSON.parse(fs.readFileSync(source, "utf8"));
|
|
401
|
+
}
|
|
402
|
+
catch (e) {
|
|
403
|
+
throw new UsageError(`cannot read ${positional}: ${e.message}`);
|
|
404
|
+
}
|
|
405
|
+
let suite;
|
|
406
|
+
try {
|
|
407
|
+
suite = evalSetToSuite(data, skill, source);
|
|
408
|
+
}
|
|
409
|
+
catch (e) {
|
|
410
|
+
if (e instanceof ConfigError)
|
|
411
|
+
throw new UsageError(`${positional}: ${e.errors.join("; ")}`);
|
|
412
|
+
throw e;
|
|
413
|
+
}
|
|
414
|
+
if (flags.out === undefined) {
|
|
415
|
+
io.stdout.write(suite.text);
|
|
416
|
+
return 0;
|
|
417
|
+
}
|
|
418
|
+
const out = path.resolve(io.cwd, flags.out);
|
|
419
|
+
if (fs.existsSync(out) && !flags.force)
|
|
420
|
+
throw new UsageError(`${flags.out} exists, use --force to overwrite`);
|
|
421
|
+
try {
|
|
422
|
+
fs.writeFileSync(out, suite.text);
|
|
423
|
+
}
|
|
424
|
+
catch (e) {
|
|
425
|
+
throw new UsageError(`cannot write ${flags.out}: ${e.message}`);
|
|
426
|
+
}
|
|
427
|
+
const total = suite.positive + suite.negative;
|
|
428
|
+
io.stdout.write(`wrote ${flags.out} (${total} cases: ${suite.positive} should load ${skill}, ${suite.negative} should not)\n`);
|
|
429
|
+
return 0;
|
|
430
|
+
}
|
|
375
431
|
const UNCOVERED_WIDTH = 78;
|
|
376
432
|
function printLint(out, report, docs, suite) {
|
|
377
433
|
const casesPart = suite ? `${report.cases} cases` : "no cases file";
|
|
@@ -473,8 +529,8 @@ async function runCommand(file, flags, io, deps) {
|
|
|
473
529
|
const startedAtMs = Date.now();
|
|
474
530
|
const fatal = new FatalStop();
|
|
475
531
|
const outcome = flags.batch
|
|
476
|
-
? await runBatched(adapter.runBatch.bind(adapter), items, flags, model,
|
|
477
|
-
: await runIndividually(adapter, items, flags, model, directive,
|
|
532
|
+
? await runBatched(adapter.runBatch.bind(adapter), items, flags, model, reporter, budget, fatal)
|
|
533
|
+
: await runIndividually(adapter, items, flags, model, directive, reporter, budget, fatal);
|
|
478
534
|
if (fatal.message !== null) {
|
|
479
535
|
// an environment problem, not a routing result: no summary, no reports
|
|
480
536
|
io.stderr.write(`skillcheck: stopped, the agent cannot run: ${fatal.message}\n`);
|
|
@@ -485,6 +541,7 @@ async function runCommand(file, flags, io, deps) {
|
|
|
485
541
|
for (const item of skipped)
|
|
486
542
|
reporter.caseSkipped(item.c);
|
|
487
543
|
const notStartedRuns = skipped.reduce((n, item) => n + item.runs.length - item.done, 0);
|
|
544
|
+
const skippedRunCosts = skipped.flatMap((item) => item.runs.flatMap((v) => (v ? [v.costUsd] : [])));
|
|
488
545
|
const ordered = outcome.results.filter((r) => r !== undefined);
|
|
489
546
|
const expected = expectedNames(selected);
|
|
490
547
|
const unavailable = outcome.sawAvailability ? expected.filter((name) => !outcome.available.has(name)) : [];
|
|
@@ -495,6 +552,7 @@ async function runCommand(file, flags, io, deps) {
|
|
|
495
552
|
skipped: skipped.length,
|
|
496
553
|
budget: skipped.length > 0 ? { limitUsd: flags.budget, spent: budget.spent, notStartedRuns } : undefined,
|
|
497
554
|
estimatedUsd: budget.spent,
|
|
555
|
+
skippedRunCosts,
|
|
498
556
|
});
|
|
499
557
|
if (flags.batch) {
|
|
500
558
|
// cases that only errored have no answer to confirm
|
|
@@ -513,7 +571,9 @@ async function runCommand(file, flags, io, deps) {
|
|
|
513
571
|
cases: items.map((item) => ({ c: item.c, threshold: item.threshold, result: outcome.results[item.jobIndex] ?? null })),
|
|
514
572
|
unavailable,
|
|
515
573
|
confusion: pairs,
|
|
574
|
+
batch: flags.batch,
|
|
516
575
|
estimatedCostUsd: budget.spent,
|
|
576
|
+
skippedRunCosts,
|
|
517
577
|
budgetUsd: flags.budget ?? null,
|
|
518
578
|
budgetReached: skipped.length > 0,
|
|
519
579
|
});
|
|
@@ -526,6 +586,8 @@ async function runCommand(file, flags, io, deps) {
|
|
|
526
586
|
}
|
|
527
587
|
if (flags.junit !== undefined)
|
|
528
588
|
writeReportFile(flags.junit, toJunit(report), "--junit");
|
|
589
|
+
if (flags.markdown !== undefined)
|
|
590
|
+
writeReportFile(flags.markdown, toMarkdown(report), "--markdown");
|
|
529
591
|
return ordered.some((r) => !r.ok) || skipped.length > 0 ? 1 : 0;
|
|
530
592
|
}
|
|
531
593
|
function writeReportFile(path, text, flag) {
|
|
@@ -537,9 +599,9 @@ function writeReportFile(path, text, flag) {
|
|
|
537
599
|
}
|
|
538
600
|
}
|
|
539
601
|
/** One agent call per case: the normal mode. */
|
|
540
|
-
async function runIndividually(adapter, items, flags, model, directive,
|
|
602
|
+
async function runIndividually(adapter, items, flags, model, directive, reporter, budget, fatal) {
|
|
541
603
|
const totalRuns = items.reduce((n, item) => n + item.runs.length, 0);
|
|
542
|
-
reporter.header(items.length,
|
|
604
|
+
reporter.header(items.length, repeatLabel(items), 1, totalRuns);
|
|
543
605
|
const jobs = items.flatMap((item, jobIndex) => item.runs.map((_, runIndex) => ({ item, runIndex, jobIndex })));
|
|
544
606
|
// keep cases in file order even though runs of different cases interleave
|
|
545
607
|
const results = new Array(items.length);
|
|
@@ -575,17 +637,19 @@ async function runIndividually(adapter, items, flags, model, directive, repeat,
|
|
|
575
637
|
}
|
|
576
638
|
/** One agent call per chunk of cases; the model states its choice per request
|
|
577
639
|
* instead of loading skills. Verdicts flow through the same judge. */
|
|
578
|
-
async function runBatched(runBatch, items, flags, model,
|
|
640
|
+
async function runBatched(runBatch, items, flags, model, reporter, budget, fatal) {
|
|
579
641
|
const chunks = [];
|
|
580
642
|
for (let i = 0; i < items.length; i += flags.batchSize)
|
|
581
643
|
chunks.push(items.slice(i, i + flags.batchSize));
|
|
582
644
|
const rounds = Math.max(...items.map((item) => item.runs.length));
|
|
583
645
|
const jobs = [];
|
|
584
646
|
for (let round = 0; round < rounds; round++) {
|
|
647
|
+
// a round past every repeat in the chunk makes no call
|
|
585
648
|
for (const chunk of chunks)
|
|
586
|
-
|
|
649
|
+
if (chunk.some((item) => round < item.runs.length))
|
|
650
|
+
jobs.push({ chunk, round });
|
|
587
651
|
}
|
|
588
|
-
reporter.batchHeader(items.length,
|
|
652
|
+
reporter.batchHeader(items.length, repeatLabel(items), jobs.length);
|
|
589
653
|
const results = new Array(items.length);
|
|
590
654
|
await runPool(jobs, flags.jobs, async (job) => {
|
|
591
655
|
// a case with a smaller repeat only takes its first rounds
|
|
@@ -597,6 +661,8 @@ async function runBatched(runBatch, items, flags, model, repeat, reporter, budge
|
|
|
597
661
|
const parsed = parseBatchAnswer(br.structured, br.text);
|
|
598
662
|
const chunkError = br.error ?? parsed.error; // a chunk error is every case's error
|
|
599
663
|
fatal.note(br.error);
|
|
664
|
+
// one call, one budget entry: its usage prices it when the cost never came
|
|
665
|
+
budget.add(br.costUsd, br.usage ?? null);
|
|
600
666
|
active.forEach((item, i) => {
|
|
601
667
|
const answer = parsed.answers.get(i + 1);
|
|
602
668
|
const r = {
|
|
@@ -609,7 +675,6 @@ async function runBatched(runBatch, items, flags, model, repeat, reporter, budge
|
|
|
609
675
|
durationMs: br.durationMs,
|
|
610
676
|
};
|
|
611
677
|
const verdict = judge(item.c, r);
|
|
612
|
-
budget.add(verdict.costUsd);
|
|
613
678
|
item.runs[job.round] = verdict;
|
|
614
679
|
item.done++;
|
|
615
680
|
if (item.done === item.runs.length) {
|
|
@@ -621,6 +686,13 @@ async function runBatched(runBatch, items, flags, model, repeat, reporter, budge
|
|
|
621
686
|
}, { shouldStart: () => fatal.canStart(budget) });
|
|
622
687
|
return { results, available: new Set(), sawAvailability: false };
|
|
623
688
|
}
|
|
689
|
+
/** "3", or "1–3" when cases override the repeat. */
|
|
690
|
+
function repeatLabel(items) {
|
|
691
|
+
const counts = items.map((item) => item.runs.length);
|
|
692
|
+
const min = Math.min(...counts);
|
|
693
|
+
const max = Math.max(...counts);
|
|
694
|
+
return min === max ? String(min) : `${min}–${max}`;
|
|
695
|
+
}
|
|
624
696
|
function selectCases(suite, agent, only) {
|
|
625
697
|
const forAgent = suite.cases.filter((c) => !c.agents || c.agents.includes(agent));
|
|
626
698
|
if (!only)
|
package/dist/import.js
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
import * as path from "node:path";
|
|
2
|
+
import { parse as parseYaml } from "yaml";
|
|
3
|
+
import { ConfigError, parseSuite } from "./cases.js";
|
|
4
|
+
const PLAIN_NAME = /^[A-Za-z0-9._:/-]+$/;
|
|
5
|
+
/**
|
|
6
|
+
* A skill-creator trigger eval set ([{query, should_trigger}], one skill) as a
|
|
7
|
+
* cases file: should_trigger true becomes expect, false becomes forbid.
|
|
8
|
+
*/
|
|
9
|
+
export function evalSetToSuite(data, skill, source) {
|
|
10
|
+
const errors = [];
|
|
11
|
+
if (!Array.isArray(data) || data.length === 0) {
|
|
12
|
+
throw new ConfigError(["expected a JSON array of {query, should_trigger}"]);
|
|
13
|
+
}
|
|
14
|
+
const items = [];
|
|
15
|
+
data.forEach((item, i) => {
|
|
16
|
+
const n = i + 1;
|
|
17
|
+
if (typeof item !== "object" || item === null || Array.isArray(item)) {
|
|
18
|
+
errors.push(`item ${n}: expected an object with query and should_trigger`);
|
|
19
|
+
return;
|
|
20
|
+
}
|
|
21
|
+
const { query, should_trigger: trigger } = item;
|
|
22
|
+
if (typeof query !== "string" || query.trim() === "")
|
|
23
|
+
errors.push(`item ${n}: query must be a non-empty string`);
|
|
24
|
+
if (typeof trigger !== "boolean")
|
|
25
|
+
errors.push(`item ${n}: should_trigger must be true or false`);
|
|
26
|
+
if (typeof query === "string" && typeof trigger === "boolean")
|
|
27
|
+
items.push({ query, trigger });
|
|
28
|
+
});
|
|
29
|
+
if (errors.length > 0)
|
|
30
|
+
throw new ConfigError(errors);
|
|
31
|
+
// plain only when YAML reads it back as the same string: not true, null, 123
|
|
32
|
+
const name = PLAIN_NAME.test(skill) && parseYaml(skill) === skill ? skill : JSON.stringify(skill);
|
|
33
|
+
const lines = [
|
|
34
|
+
`# skillcheck cases imported from ${path.basename(source)} (skill-creator trigger eval set).`,
|
|
35
|
+
"# should_trigger: true became expect, false became forbid: another skill may",
|
|
36
|
+
`# still load there, only ${skill} must not.`,
|
|
37
|
+
"agent: claude",
|
|
38
|
+
"repeat: 1",
|
|
39
|
+
"threshold: 1.0",
|
|
40
|
+
"cases:",
|
|
41
|
+
];
|
|
42
|
+
for (const { query, trigger } of items) {
|
|
43
|
+
// a JSON string is a valid YAML double-quoted scalar, newlines and quotes included
|
|
44
|
+
lines.push(` - query: ${JSON.stringify(query)}`, ` ${trigger ? "expect" : "forbid"}: [${name}]`);
|
|
45
|
+
}
|
|
46
|
+
const text = lines.join("\n") + "\n";
|
|
47
|
+
parseSuite(parseYaml(text));
|
|
48
|
+
const positive = items.filter((i) => i.trigger).length;
|
|
49
|
+
return { text, positive, negative: items.length - positive };
|
|
50
|
+
}
|
package/dist/markdown.js
ADDED
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
/** An HTML comment so GitHub renders the report body clean under it. */
|
|
2
|
+
export const MARKDOWN_MARKER = "<!-- skillcheck -->";
|
|
3
|
+
/** Failure-focused Markdown summary for PR comments and GitHub job summaries. */
|
|
4
|
+
export function toMarkdown(report) {
|
|
5
|
+
const s = report.summary;
|
|
6
|
+
const lines = [MARKDOWN_MARKER, header(report)];
|
|
7
|
+
lines.push("");
|
|
8
|
+
lines.push(metaLine(report));
|
|
9
|
+
if (report.batch) {
|
|
10
|
+
lines.push("");
|
|
11
|
+
lines.push("> Batch mode: answers are the model's stated choice, not an actual Skill call. Confirm failures with a normal run.");
|
|
12
|
+
}
|
|
13
|
+
if (s.budgetReached && s.budgetUsd !== null) {
|
|
14
|
+
lines.push("");
|
|
15
|
+
lines.push(`> ⚠️ Budget $${s.budgetUsd.toFixed(2)} reached, remaining cases were skipped.`);
|
|
16
|
+
}
|
|
17
|
+
if (report.unavailable.length > 0) {
|
|
18
|
+
lines.push("");
|
|
19
|
+
lines.push(`> ⚠️ Expected skills not available to the agent: ${report.unavailable.join(", ")}`);
|
|
20
|
+
}
|
|
21
|
+
if (s.failed + s.skipped > 0) {
|
|
22
|
+
lines.push("");
|
|
23
|
+
lines.push("| | Case | Request | Loaded | Reason |");
|
|
24
|
+
lines.push("|---|---|---|---|---|");
|
|
25
|
+
for (const c of report.cases) {
|
|
26
|
+
if (c.status === "passed")
|
|
27
|
+
continue;
|
|
28
|
+
lines.push(failedRow(c));
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
const passed = report.cases.filter((c) => c.status === "passed");
|
|
32
|
+
if (passed.length > 0) {
|
|
33
|
+
lines.push("");
|
|
34
|
+
lines.push(`<details><summary>Passed (${passed.length})</summary>`);
|
|
35
|
+
lines.push("");
|
|
36
|
+
lines.push("| Case | Request | Loaded |");
|
|
37
|
+
lines.push("|---|---|---|");
|
|
38
|
+
for (const c of passed)
|
|
39
|
+
lines.push(passedRow(c));
|
|
40
|
+
lines.push("");
|
|
41
|
+
lines.push("</details>");
|
|
42
|
+
}
|
|
43
|
+
if (report.confusion.length > 0) {
|
|
44
|
+
lines.push("");
|
|
45
|
+
lines.push("**Confusion** — descriptions to rewrite:");
|
|
46
|
+
lines.push("");
|
|
47
|
+
for (const p of report.confusion)
|
|
48
|
+
lines.push(`- expected \`${p.expected}\` → got \`${p.got}\` (${p.count})`);
|
|
49
|
+
}
|
|
50
|
+
return lines.join("\n") + "\n";
|
|
51
|
+
}
|
|
52
|
+
/** ✅ with the pass count alone, ❌ with failed and skipped counts. */
|
|
53
|
+
function header(report) {
|
|
54
|
+
const s = report.summary;
|
|
55
|
+
if (s.failed === 0 && s.skipped === 0) {
|
|
56
|
+
return `### ✅ skillcheck: ${s.cases} passed`;
|
|
57
|
+
}
|
|
58
|
+
const failedPart = s.failed > 0 ? `${s.failed} failed` : null;
|
|
59
|
+
const skippedPart = s.skipped > 0 ? `${s.skipped} skipped` : null;
|
|
60
|
+
const parts = [failedPart, skippedPart].filter((p) => p !== null).join(", ");
|
|
61
|
+
return `### ❌ skillcheck: ${parts} of ${s.cases}`;
|
|
62
|
+
}
|
|
63
|
+
/** Model, batch flag, runs, cost and duration, " · " separated. */
|
|
64
|
+
function metaLine(report) {
|
|
65
|
+
const s = report.summary;
|
|
66
|
+
const parts = [];
|
|
67
|
+
if (report.model !== null)
|
|
68
|
+
parts.push(report.model);
|
|
69
|
+
if (report.batch)
|
|
70
|
+
parts.push("batch mode");
|
|
71
|
+
parts.push(`${s.runs} run${s.runs === 1 ? "" : "s"}`);
|
|
72
|
+
if (s.unknownCostRuns > 0 && s.estimatedCostUsd > s.costUsd)
|
|
73
|
+
parts.push(`cost ~$${s.estimatedCostUsd.toFixed(2)}`);
|
|
74
|
+
else
|
|
75
|
+
parts.push(`cost $${s.costUsd.toFixed(2)}`);
|
|
76
|
+
parts.push(duration(report.durationMs));
|
|
77
|
+
return parts.join(" · ");
|
|
78
|
+
}
|
|
79
|
+
/** Below a minute in seconds, then minutes and seconds, seconds rounded down. */
|
|
80
|
+
function duration(ms) {
|
|
81
|
+
const total = Math.floor(ms / 1000);
|
|
82
|
+
if (total < 60)
|
|
83
|
+
return `${total}s`;
|
|
84
|
+
return `${Math.floor(total / 60)}m ${total % 60}s`;
|
|
85
|
+
}
|
|
86
|
+
/** ⏭️ for skipped, ⚠️ when every failed run errored, ❌ for the rest. */
|
|
87
|
+
function failedRow(c) {
|
|
88
|
+
const cells = c.status === "skipped"
|
|
89
|
+
? ["⏭️", caseLabel(c), oneLine(c.query), loaded(c), "budget reached"]
|
|
90
|
+
: [failMark(c), caseLabel(c), oneLine(c.query), loaded(c), reasonCell(c)];
|
|
91
|
+
return `| ${cells.map(esc).join(" | ")} |`;
|
|
92
|
+
}
|
|
93
|
+
/** ⚠️ when every non-ok run errored (the model never answered), ❌ otherwise. */
|
|
94
|
+
function failMark(c) {
|
|
95
|
+
const bad = c.runs.filter((r) => !r.ok);
|
|
96
|
+
return bad.length > 0 && bad.every((r) => r.error !== null) ? "⚠️" : "❌";
|
|
97
|
+
}
|
|
98
|
+
/** The first non-ok run's reason, the pass share prefixed over several runs,
|
|
99
|
+
* a diagnosis appended in italics, then the case note. */
|
|
100
|
+
function reasonCell(c) {
|
|
101
|
+
const bad = c.runs.filter((r) => !r.ok);
|
|
102
|
+
const parts = c.runs.length > 1 ? [`${c.passed}/${c.runs.length}`, bad[0]?.reason ?? ""] : [bad[0]?.reason ?? ""];
|
|
103
|
+
let text = parts.join(" · ");
|
|
104
|
+
const diagnosis = bad.find((r) => r.diagnosis !== null)?.diagnosis;
|
|
105
|
+
if (diagnosis)
|
|
106
|
+
text += ` — _${diagnosis}_`;
|
|
107
|
+
if (c.note !== null)
|
|
108
|
+
text += ` · note: ${c.note}`;
|
|
109
|
+
return text;
|
|
110
|
+
}
|
|
111
|
+
function passedRow(c) {
|
|
112
|
+
const score = c.runs.length > 1 ? ` (${c.passed}/${c.runs.length})` : "";
|
|
113
|
+
const cells = [`${caseLabel(c)}${score}`, oneLine(c.query), loaded(c)];
|
|
114
|
+
return `| ${cells.map(esc).join(" | ")} |`;
|
|
115
|
+
}
|
|
116
|
+
function caseLabel(c) {
|
|
117
|
+
return `#${c.id ?? c.index}`;
|
|
118
|
+
}
|
|
119
|
+
/** Unique loaded skills across runs, "—" when none. */
|
|
120
|
+
function loaded(c) {
|
|
121
|
+
const names = [...new Set(c.runs.flatMap((r) => r.loaded))];
|
|
122
|
+
return names.length > 0 ? names.join(", ") : "—";
|
|
123
|
+
}
|
|
124
|
+
/** Collapse to one line, truncated to 80 chars with "…". */
|
|
125
|
+
function oneLine(text) {
|
|
126
|
+
const flat = text.replace(/\s+/g, " ").trim();
|
|
127
|
+
return flat.length > 80 ? `${flat.slice(0, 80)}…` : flat;
|
|
128
|
+
}
|
|
129
|
+
/** Pipes so they cannot break the table, newlines as spaces, HTML inert. */
|
|
130
|
+
function esc(text) {
|
|
131
|
+
return text
|
|
132
|
+
.replace(/\|/g, "\\|")
|
|
133
|
+
.replace(/[\r\n]+/g, " ")
|
|
134
|
+
.replace(/</g, "<");
|
|
135
|
+
}
|
package/dist/report.js
CHANGED
|
@@ -11,7 +11,7 @@ export class Reporter {
|
|
|
11
11
|
this.useColor = out.isTTY === true && !process.env.NO_COLOR;
|
|
12
12
|
}
|
|
13
13
|
header(nCases, repeat, nAgents, totalRuns) {
|
|
14
|
-
const runs = totalRuns ?? nCases * repeat * nAgents;
|
|
14
|
+
const runs = totalRuns ?? nCases * Number(repeat) * nAgents;
|
|
15
15
|
const agents = nAgents === 1 ? "agent" : "agents";
|
|
16
16
|
this.write(`${nCases} cases × ${repeat} repeat × ${nAgents} ${agents} = ${runs} runs\n`);
|
|
17
17
|
}
|
|
@@ -36,6 +36,8 @@ export class Reporter {
|
|
|
36
36
|
if (diagnosed?.diagnosis) {
|
|
37
37
|
this.write(` ${this.paint("33", `diagnosis ${label}: ${diagnosed.diagnosis}`)}\n`);
|
|
38
38
|
}
|
|
39
|
+
if (!res.ok && res.case.note)
|
|
40
|
+
this.write(` ${this.paint("2", `note ${label}: ${oneLine(res.case.note)}`)}\n`);
|
|
39
41
|
}
|
|
40
42
|
/** A case the budget never let start; printed after all real runs settle. */
|
|
41
43
|
caseSkipped(c) {
|
|
@@ -68,7 +70,8 @@ export class Reporter {
|
|
|
68
70
|
const failed = results.filter((r) => !r.ok).length;
|
|
69
71
|
const skipped = extra?.skipped ?? 0;
|
|
70
72
|
const skippedPart = skipped > 0 ? `, ${skipped} skipped` : "";
|
|
71
|
-
|
|
73
|
+
const costs = [...verdicts.map((v) => v.costUsd), ...(extra?.skippedRunCosts ?? [])];
|
|
74
|
+
this.write(`${failed} failed${skippedPart} of ${results.length + skipped} · runs ${costs.length} · ${costLine(costs, extra?.estimatedUsd)}\n`);
|
|
72
75
|
}
|
|
73
76
|
paint(code, text) {
|
|
74
77
|
return this.useColor ? `\u001b[${code}m${text}${RESET}` : text;
|
package/dist/results.js
CHANGED
|
@@ -2,7 +2,7 @@ import { readVersion } from "./version.js";
|
|
|
2
2
|
/** The single source of truth for --json and --junit; computed once per run. */
|
|
3
3
|
export function buildReport(input) {
|
|
4
4
|
const verdicts = input.cases.flatMap((e) => e.result?.runs ?? []);
|
|
5
|
-
const costs = verdicts.map((v) => v.costUsd);
|
|
5
|
+
const costs = [...verdicts.map((v) => v.costUsd), ...(input.skippedRunCosts ?? [])];
|
|
6
6
|
const known = costs.filter((c) => c !== null);
|
|
7
7
|
return {
|
|
8
8
|
tool: "skillcheck",
|
|
@@ -10,13 +10,14 @@ export function buildReport(input) {
|
|
|
10
10
|
file: input.file,
|
|
11
11
|
agent: input.agent,
|
|
12
12
|
model: input.model,
|
|
13
|
+
batch: input.batch ?? false,
|
|
13
14
|
startedAt: new Date(input.startedAtMs).toISOString(),
|
|
14
15
|
durationMs: input.durationMs,
|
|
15
16
|
summary: {
|
|
16
17
|
cases: input.cases.length,
|
|
17
18
|
failed: input.cases.filter((e) => e.result && !e.result.ok).length,
|
|
18
19
|
skipped: input.cases.filter((e) => !e.result).length,
|
|
19
|
-
runs:
|
|
20
|
+
runs: costs.length,
|
|
20
21
|
costUsd: known.reduce((a, b) => a + b, 0),
|
|
21
22
|
unknownCostRuns: costs.length - known.length,
|
|
22
23
|
estimatedCostUsd: input.estimatedCostUsd,
|