@uinaf/skillcheck 1.2.0 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/cli.js +89 -36
- package/dist/grok-provider.js +0 -1
- package/dist/scenario.js +51 -17
- package/dist/skill-evidence.js +49 -0
- package/dist/transform.js +3 -3
- package/docs/adoption.md +0 -1
- package/docs/scenarios.md +30 -5
- package/docs/usage.md +22 -7
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -45,7 +45,7 @@ current directory.
|
|
|
45
45
|
<root>/skills/<skill>/SKILL.md linted
|
|
46
46
|
<root>/skills/<skill>/evals/<scenario>/task.md the problem + input files
|
|
47
47
|
<root>/skills/<skill>/evals/<scenario>/criteria.json the weighted checklist
|
|
48
|
-
<root>/.skillcheck/ results,
|
|
48
|
+
<root>/.skillcheck/ results, scorecards
|
|
49
49
|
```
|
|
50
50
|
|
|
51
51
|
`cli/*/skills/<skill>/` is scanned too, for repos that keep a skill next to the
|
package/dist/cli.js
CHANGED
|
@@ -2,7 +2,9 @@
|
|
|
2
2
|
import { lintSkills } from "./lint.js";
|
|
3
3
|
import { encodeRunNamePart, generateRun, requiredEvalPackages, resolvePackageDir, runNameFor } from "./scenario.js";
|
|
4
4
|
import { execFileSync, spawnSync } from "node:child_process";
|
|
5
|
+
import { createHash } from "node:crypto";
|
|
5
6
|
import fs from "node:fs";
|
|
7
|
+
import os from "node:os";
|
|
6
8
|
import path from "node:path";
|
|
7
9
|
import { fileURLToPath } from "node:url";
|
|
8
10
|
//#region src/cli.ts
|
|
@@ -48,7 +50,7 @@ function parseArgs(argv) {
|
|
|
48
50
|
const v = argv[++i];
|
|
49
51
|
if (v === void 0 || v.startsWith("--")) throw new Error(`${a} needs a value`);
|
|
50
52
|
flags.set(a, v);
|
|
51
|
-
} else if (a === "--all" || a === "--allow-mixed") flags.set(a, true);
|
|
53
|
+
} else if (a === "--all" || a === "--allow-mixed" || a === "--control") flags.set(a, true);
|
|
52
54
|
else throw new Error(`unknown flag: ${a}`);
|
|
53
55
|
}
|
|
54
56
|
return {
|
|
@@ -61,9 +63,10 @@ function resolveRoot(flags) {
|
|
|
61
63
|
}
|
|
62
64
|
function stateDirs(root) {
|
|
63
65
|
const base = path.join(root, ".skillcheck");
|
|
66
|
+
const id = createHash("sha256").update(path.resolve(root)).digest("hex").slice(0, 12);
|
|
64
67
|
return {
|
|
65
68
|
results: path.join(base, "results"),
|
|
66
|
-
scratch: path.join(
|
|
69
|
+
scratch: path.join(fs.realpathSync(os.tmpdir()), `skillcheck-${id}`),
|
|
67
70
|
scorecards: path.join(base, "scorecards")
|
|
68
71
|
};
|
|
69
72
|
}
|
|
@@ -95,6 +98,7 @@ function runOptions(flags) {
|
|
|
95
98
|
judgeModel,
|
|
96
99
|
judgeEffort,
|
|
97
100
|
maxTurns: flags.has("--max-turns") ? parsePositiveInt("--max-turns", flags.get("--max-turns")) : void 0,
|
|
101
|
+
control: flags.get("--control") === true,
|
|
98
102
|
trials: flags.has("--trials") ? parsePositiveInt("--trials", flags.get("--trials")) : 1
|
|
99
103
|
};
|
|
100
104
|
}
|
|
@@ -104,7 +108,8 @@ function runConfigOf(opts) {
|
|
|
104
108
|
agent_effort: opts.agentEffort ?? null,
|
|
105
109
|
judge_model: opts.judgeModel,
|
|
106
110
|
judge_effort: opts.judgeEffort ?? null,
|
|
107
|
-
trials: opts.trials ?? 1
|
|
111
|
+
trials: opts.trials ?? 1,
|
|
112
|
+
agent_access: "online"
|
|
108
113
|
};
|
|
109
114
|
}
|
|
110
115
|
function configKey(c) {
|
|
@@ -113,12 +118,13 @@ function configKey(c) {
|
|
|
113
118
|
c.agent_effort ?? null,
|
|
114
119
|
c.judge_model,
|
|
115
120
|
c.judge_effort ?? null,
|
|
116
|
-
c.trials ?? 1
|
|
121
|
+
c.trials ?? 1,
|
|
122
|
+
c.agent_access ?? "offline"
|
|
117
123
|
]);
|
|
118
124
|
}
|
|
119
125
|
function describeConfig(c) {
|
|
120
126
|
const effort = (e) => e ? `@${e}` : "";
|
|
121
|
-
return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}`;
|
|
127
|
+
return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}, ${c.agent_access ?? "offline"}`;
|
|
122
128
|
}
|
|
123
129
|
function assertUniformConfig(entries, allowMixed) {
|
|
124
130
|
if (allowMixed) return;
|
|
@@ -180,12 +186,12 @@ function classifyRow(raw, stats) {
|
|
|
180
186
|
if (typeof res.score !== "number" || typeof res.success !== "boolean") return { error: "promptfoo result carried no usable score" };
|
|
181
187
|
const components = Array.isArray(res.gradingResult?.componentResults) ? res.gradingResult.componentResults : [];
|
|
182
188
|
const checklist = components.find((c) => c?.metadata?.assertionSet?.type === "assert-set");
|
|
183
|
-
const skillUsed = components.find((c) => c?.assertion?.type === "skill-used");
|
|
184
|
-
if (typeof checklist?.score !== "number" || typeof skillUsed?.
|
|
189
|
+
const skillUsed = components.find((c) => c?.assertion?.type === "skill-used" || c?.assertion?.metric === "skill-used");
|
|
190
|
+
if (typeof checklist?.score !== "number" || typeof skillUsed?.score !== "number") return { error: "promptfoo result carried no checklist or skill-used verdict" };
|
|
185
191
|
return {
|
|
186
192
|
score: checklist.score,
|
|
187
193
|
pass: res.success,
|
|
188
|
-
skillUsed: skillUsed.
|
|
194
|
+
skillUsed: skillUsed.score >= 1
|
|
189
195
|
};
|
|
190
196
|
}
|
|
191
197
|
const NOISY_SPREAD = .2;
|
|
@@ -235,19 +241,32 @@ function metaPath(resultPath) {
|
|
|
235
241
|
function attemptPath(resultPath) {
|
|
236
242
|
return `${resultPath}.attempt`;
|
|
237
243
|
}
|
|
244
|
+
function ensurePrivateDir(dir) {
|
|
245
|
+
fs.mkdirSync(dir, {
|
|
246
|
+
recursive: true,
|
|
247
|
+
mode: 448
|
|
248
|
+
});
|
|
249
|
+
const st = fs.lstatSync(dir);
|
|
250
|
+
const uid = process.getuid?.();
|
|
251
|
+
if (!st.isDirectory() || uid !== void 0 && st.uid !== uid) throw new Error(`scratch dir ${dir} is not a directory owned by this user`);
|
|
252
|
+
if ((st.mode & 63) !== 0) fs.chmodSync(dir, 448);
|
|
253
|
+
}
|
|
238
254
|
function runScenario(scenarioDir, opts, root) {
|
|
239
255
|
const dirs = stateDirs(root);
|
|
256
|
+
ensurePrivateDir(dirs.scratch);
|
|
240
257
|
const { name, configPath, skill, scenario } = generateRun(path.resolve(scenarioDir), opts, {
|
|
241
258
|
scratchDir: dirs.scratch,
|
|
242
259
|
transformPath: path.join(here, `transform${selfExt}`),
|
|
243
|
-
grokProviderPath: path.join(here, `grok-provider${selfExt}`)
|
|
260
|
+
grokProviderPath: path.join(here, `grok-provider${selfExt}`),
|
|
261
|
+
skillEvidencePath: path.join(here, `skill-evidence${selfExt}`)
|
|
244
262
|
});
|
|
245
263
|
fs.mkdirSync(dirs.results, { recursive: true });
|
|
246
264
|
const resultPath = path.join(dirs.results, `${name}.json`);
|
|
247
265
|
const identity = {
|
|
248
266
|
skill,
|
|
249
267
|
scenario,
|
|
250
|
-
harness: opts.harness
|
|
268
|
+
harness: opts.harness,
|
|
269
|
+
variant: opts.control ? "control" : "skill"
|
|
251
270
|
};
|
|
252
271
|
fs.writeFileSync(attemptPath(resultPath), JSON.stringify(identity) + "\n");
|
|
253
272
|
fs.rmSync(resultPath, { force: true });
|
|
@@ -314,7 +333,8 @@ function resultRunConfig(raw, meta, harness) {
|
|
|
314
333
|
agent_effort: m.agent_effort ?? null,
|
|
315
334
|
judge_model: m.judge_model,
|
|
316
335
|
judge_effort: m.judge_effort ?? null,
|
|
317
|
-
trials: m.trials ?? 1
|
|
336
|
+
trials: m.trials ?? 1,
|
|
337
|
+
agent_access: m.agent_access === "online" ? "online" : "offline"
|
|
318
338
|
};
|
|
319
339
|
const r = raw;
|
|
320
340
|
const agent = r?.config?.providers?.[0]?.config;
|
|
@@ -326,7 +346,8 @@ function resultRunConfig(raw, meta, harness) {
|
|
|
326
346
|
agent_effort: typeof agent?.effort === "string" ? agent.effort : null,
|
|
327
347
|
judge_model: judgeName(judge),
|
|
328
348
|
judge_effort: typeof judgeEffort === "string" ? judgeEffort : null,
|
|
329
|
-
trials: r?.results?.results?.length ?? 1
|
|
349
|
+
trials: r?.results?.results?.length ?? 1,
|
|
350
|
+
agent_access: "offline"
|
|
330
351
|
};
|
|
331
352
|
}
|
|
332
353
|
function readJson(file) {
|
|
@@ -356,7 +377,7 @@ function discoverScenarios(root) {
|
|
|
356
377
|
}
|
|
357
378
|
return found.sort();
|
|
358
379
|
}
|
|
359
|
-
const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N]";
|
|
380
|
+
const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N] [--control]";
|
|
360
381
|
function cmdRun(argv) {
|
|
361
382
|
const { positional, flags } = parseArgs(argv);
|
|
362
383
|
if (positional.length !== 1) fail(`usage: skillcheck run <scenario-dir> ${RUN_FLAGS}`);
|
|
@@ -381,7 +402,7 @@ function cmdSweep(argv) {
|
|
|
381
402
|
const wanted = configKey(runConfigOf(opts));
|
|
382
403
|
let passed = 0, failed = 0, errored = 0, skipped = 0;
|
|
383
404
|
for (const dir of discoverScenarios(root)) {
|
|
384
|
-
const name = runNameFor(dir, opts.harness);
|
|
405
|
+
const name = runNameFor(dir, opts.harness, opts.control);
|
|
385
406
|
const resultPath = path.join(resultsDir, `${name}.json`);
|
|
386
407
|
if (!all && fs.existsSync(resultPath) && !fs.existsSync(attemptPath(resultPath))) {
|
|
387
408
|
const raw = readJson(resultPath);
|
|
@@ -416,7 +437,8 @@ function resultIdentity(file, dir) {
|
|
|
416
437
|
if (identity !== null && typeof identity === "object" && "skill" in identity && typeof identity.skill === "string" && "scenario" in identity && typeof identity.scenario === "string" && "harness" in identity && (identity.harness === "claude" || identity.harness === "codex" || identity.harness === "grok")) return {
|
|
417
438
|
skill: identity.skill,
|
|
418
439
|
scenario: identity.scenario,
|
|
419
|
-
harness: identity.harness
|
|
440
|
+
harness: identity.harness,
|
|
441
|
+
variant: "variant" in identity && identity.variant === "control" ? "control" : "skill"
|
|
420
442
|
};
|
|
421
443
|
} catch {}
|
|
422
444
|
if (/^~v3~[0-9a-f]{64}$/.test(base)) throw new Error(`missing identity metadata for ${file}`);
|
|
@@ -429,13 +451,17 @@ function resultIdentity(file, dir) {
|
|
|
429
451
|
return part;
|
|
430
452
|
}
|
|
431
453
|
};
|
|
432
|
-
const
|
|
454
|
+
const unsuffixed = base.replace(/--control$/, "");
|
|
455
|
+
const variant = unsuffixed !== base && unsuffixed.includes("--") ? "control" : "skill";
|
|
456
|
+
const stem = variant === "control" ? unsuffixed : base;
|
|
457
|
+
const suffix = stem.match(/--(codex|grok|cursor)$/);
|
|
433
458
|
const harness = suffix === null ? "claude" : suffix[1];
|
|
434
|
-
const [skill, ...rest] =
|
|
459
|
+
const [skill, ...rest] = stem.replace(/--(codex|grok|cursor)$/, "").split("--");
|
|
435
460
|
return {
|
|
436
461
|
skill: decode(skill),
|
|
437
462
|
scenario: decode(rest.join("--")),
|
|
438
|
-
harness
|
|
463
|
+
harness,
|
|
464
|
+
variant
|
|
439
465
|
};
|
|
440
466
|
}
|
|
441
467
|
function reduceResults(dir, allowMixed) {
|
|
@@ -460,7 +486,7 @@ function reduceResults(dir, allowMixed) {
|
|
|
460
486
|
const raw = readJson(path.join(dir, f));
|
|
461
487
|
const base = f.replace(/\.json$/, "");
|
|
462
488
|
const meta = readJson(path.join(dir, `${base}.meta.json`));
|
|
463
|
-
const { skill, scenario, harness } = resultIdentity(f, dir);
|
|
489
|
+
const { skill, scenario, harness, variant } = resultIdentity(f, dir);
|
|
464
490
|
const config = resultRunConfig(raw, meta, harness);
|
|
465
491
|
const verdict = classifyResult(raw, config.trials);
|
|
466
492
|
if ("error" in verdict) {
|
|
@@ -472,7 +498,8 @@ function reduceResults(dir, allowMixed) {
|
|
|
472
498
|
const key = entryKey({
|
|
473
499
|
skill,
|
|
474
500
|
scenario,
|
|
475
|
-
harness
|
|
501
|
+
harness,
|
|
502
|
+
variant
|
|
476
503
|
});
|
|
477
504
|
gradedAt.set(key, Math.max(gradedAt.get(key) ?? 0, fs.statSync(path.join(dir, f)).mtimeMs));
|
|
478
505
|
const sha = typeof meta?.skills_tree_sha === "string" ? meta.skills_tree_sha : "unattested";
|
|
@@ -482,6 +509,7 @@ function reduceResults(dir, allowMixed) {
|
|
|
482
509
|
skill,
|
|
483
510
|
scenario,
|
|
484
511
|
harness,
|
|
512
|
+
variant,
|
|
485
513
|
skills_tree_sha: sha,
|
|
486
514
|
...stats,
|
|
487
515
|
...config,
|
|
@@ -501,7 +529,8 @@ function entryKey(e) {
|
|
|
501
529
|
return [
|
|
502
530
|
e.skill,
|
|
503
531
|
e.scenario,
|
|
504
|
-
e.harness
|
|
532
|
+
e.harness,
|
|
533
|
+
e.variant ?? "skill"
|
|
505
534
|
].join("\0");
|
|
506
535
|
}
|
|
507
536
|
function mergeScorecard(existing, fresh) {
|
|
@@ -558,27 +587,47 @@ function cmdSummarize(argv) {
|
|
|
558
587
|
scenarios: merged.entries
|
|
559
588
|
};
|
|
560
589
|
fs.writeFileSync(out, JSON.stringify(scorecard, null, 2) + "\n");
|
|
561
|
-
|
|
590
|
+
const scenarios = merged.entries.filter((e) => e.variant !== "control");
|
|
591
|
+
console.log(`${out}: ${scenarios.length} scenario(s), ${scenarios.filter((e) => e.pass).length} passing, ${merged.entries.length - scenarios.length} control(s), ${skipped.length} skipped file(s)`);
|
|
562
592
|
if (existing.length > 0) console.log(`merged into today's scorecard: ${entries.length} from this run, ${merged.carried} carried over`);
|
|
563
593
|
console.log(`\n${formatSkillTable(summarizeSkills(merged.entries))}`);
|
|
564
|
-
for (const e of merged.entries.filter((x) => x.noisy))
|
|
594
|
+
for (const e of merged.entries.filter((x) => x.noisy)) {
|
|
595
|
+
const tag = e.variant === "control" ? ", control" : "";
|
|
596
|
+
console.log(`NOISY ${e.skill}/${e.scenario} (${e.harness}${tag}) ${formatStats(e)}`);
|
|
597
|
+
}
|
|
598
|
+
for (const s of summarizeSkills(merged.entries)) for (const scenario of s.no_lift) console.log(`NO LIFT ${s.skill}/${scenario} (${s.harness}): passes without the skill`);
|
|
565
599
|
}
|
|
566
600
|
function summarizeSkills(entries) {
|
|
567
601
|
const groups = /* @__PURE__ */ new Map();
|
|
602
|
+
const controls = /* @__PURE__ */ new Map();
|
|
568
603
|
for (const e of entries) {
|
|
569
604
|
const key = `${e.skill}\0${e.harness}`;
|
|
570
|
-
|
|
605
|
+
if (e.variant === "control") controls.set(`${key}\0${e.scenario}`, e);
|
|
606
|
+
else groups.set(key, [...groups.get(key) ?? [], e]);
|
|
571
607
|
}
|
|
572
608
|
const mean = (xs) => round4(xs.reduce((a, b) => a + b, 0) / xs.length);
|
|
573
|
-
return [...groups.
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
609
|
+
return [...groups.entries()].map(([key, rows]) => {
|
|
610
|
+
const paired = rows.flatMap((r) => {
|
|
611
|
+
const c = controls.get(`${key}\0${r.scenario}`);
|
|
612
|
+
return c === void 0 ? [] : [{
|
|
613
|
+
skill: r,
|
|
614
|
+
control: c
|
|
615
|
+
}];
|
|
616
|
+
});
|
|
617
|
+
const controlScore = paired.length === 0 ? null : mean(paired.map((p) => p.control.score));
|
|
618
|
+
return {
|
|
619
|
+
skill: rows[0].skill,
|
|
620
|
+
harness: rows[0].harness,
|
|
621
|
+
scenarios: rows.length,
|
|
622
|
+
pass_all: rows.filter((r) => r.pass).length,
|
|
623
|
+
pass_rate: mean(rows.map((r) => r.pass_rate ?? (r.pass ? 1 : 0))),
|
|
624
|
+
score: mean(rows.map((r) => r.score)),
|
|
625
|
+
noisy: rows.filter((r) => r.noisy).map((r) => r.scenario),
|
|
626
|
+
control_score: controlScore,
|
|
627
|
+
lift: controlScore === null ? null : round4(mean(paired.map((p) => p.skill.score)) - controlScore),
|
|
628
|
+
no_lift: paired.filter((p) => p.control.pass).map((p) => p.skill.scenario)
|
|
629
|
+
};
|
|
630
|
+
});
|
|
582
631
|
}
|
|
583
632
|
function formatSkillTable(rows) {
|
|
584
633
|
const table = [[
|
|
@@ -588,7 +637,9 @@ function formatSkillTable(rows) {
|
|
|
588
637
|
"pass^k",
|
|
589
638
|
"pass rate",
|
|
590
639
|
"score",
|
|
591
|
-
"noisy"
|
|
640
|
+
"noisy",
|
|
641
|
+
"control",
|
|
642
|
+
"lift"
|
|
592
643
|
], ...rows.map((r) => [
|
|
593
644
|
r.skill,
|
|
594
645
|
r.harness,
|
|
@@ -596,7 +647,9 @@ function formatSkillTable(rows) {
|
|
|
596
647
|
`${r.pass_all}/${r.scenarios}`,
|
|
597
648
|
r.pass_rate.toFixed(2),
|
|
598
649
|
r.score.toFixed(2),
|
|
599
|
-
String(r.noisy.length)
|
|
650
|
+
String(r.noisy.length),
|
|
651
|
+
r.control_score === null ? "-" : r.control_score.toFixed(2),
|
|
652
|
+
r.lift === null ? "-" : `${r.lift >= 0 ? "+" : ""}${r.lift.toFixed(2)}`
|
|
600
653
|
])];
|
|
601
654
|
const widths = table[0].map((_, i) => Math.max(...table.map((row) => row[i].length)));
|
|
602
655
|
return table.map((row) => row.map((c, i) => c.padEnd(widths[i])).join(" ").trimEnd()).join("\n");
|
|
@@ -635,4 +688,4 @@ if (isMainModule()) {
|
|
|
635
688
|
}
|
|
636
689
|
}
|
|
637
690
|
//#endregion
|
|
638
|
-
export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, formatStats, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
|
|
691
|
+
export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, ensurePrivateDir, formatStats, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
|
package/dist/grok-provider.js
CHANGED
package/dist/scenario.js
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { createRequire } from "node:module";
|
|
2
|
-
import fs from "node:fs";
|
|
3
|
-
import path from "node:path";
|
|
4
2
|
import { createHash } from "node:crypto";
|
|
3
|
+
import fs from "node:fs";
|
|
5
4
|
import os from "node:os";
|
|
5
|
+
import path from "node:path";
|
|
6
6
|
//#region src/scenario.ts
|
|
7
7
|
const DEFAULT_CLAUDE_AGENT = "claude-opus-5";
|
|
8
8
|
function loadScenario(scenarioDir) {
|
|
@@ -12,15 +12,17 @@ function loadScenario(scenarioDir) {
|
|
|
12
12
|
const taskMd = fs.readFileSync(path.join(scenarioDir, "task.md"), "utf8");
|
|
13
13
|
const criteria = JSON.parse(fs.readFileSync(path.join(scenarioDir, "criteria.json"), "utf8"));
|
|
14
14
|
if (criteria.type !== "weighted_checklist" || !Array.isArray(criteria.checklist) || criteria.checklist.length === 0) throw new Error(`unsupported or empty criteria in ${scenarioDir}`);
|
|
15
|
+
if (criteria.skill_use !== void 0 && criteria.skill_use !== "required" && criteria.skill_use !== "optional") throw new Error(`skill_use must be "required" or "optional" in ${scenarioDir}`);
|
|
15
16
|
for (const item of criteria.checklist) if (!(typeof item?.name === "string" && item.name.trim() !== "" && typeof item?.description === "string" && item.description.trim() !== "" && Number.isFinite(item?.max_score) && item.max_score > 0)) throw new Error(`invalid checklist item in ${scenarioDir}: ${JSON.stringify(item)}`);
|
|
16
17
|
const files = [];
|
|
17
|
-
|
|
18
|
+
const task = taskMd.replace(/^=+ FILE: (.+?) =+\n([\s\S]*?)\n=+ END FILE =+$/gm, (_, name, content) => {
|
|
18
19
|
files.push({
|
|
19
20
|
name: name.trim(),
|
|
20
21
|
content: content + "\n"
|
|
21
22
|
});
|
|
22
23
|
return `(Input file \`${name.trim()}\` is available in your working directory.)`;
|
|
23
24
|
});
|
|
25
|
+
let prompt = task;
|
|
24
26
|
const skillDir = path.resolve(scenarioDir, "../..");
|
|
25
27
|
if (isHiddenSkill(fs.readFileSync(path.join(skillDir, "SKILL.md"), "utf8"))) prompt = `Use the ${skill} skill for this task.\n\n${prompt}`;
|
|
26
28
|
return {
|
|
@@ -29,6 +31,7 @@ function loadScenario(scenarioDir) {
|
|
|
29
31
|
name: `${skill}--${scenario}`,
|
|
30
32
|
skillDir,
|
|
31
33
|
prompt,
|
|
34
|
+
task,
|
|
32
35
|
files,
|
|
33
36
|
criteria
|
|
34
37
|
};
|
|
@@ -37,13 +40,18 @@ function encodeRunNamePart(part) {
|
|
|
37
40
|
if (!part.includes("--") && !part.startsWith("-") && !part.endsWith("-") && !part.startsWith("~v2~")) return part;
|
|
38
41
|
return `~v2~${encodeURIComponent(part).replaceAll("-", "%2D").replaceAll("~", "%7E")}`;
|
|
39
42
|
}
|
|
40
|
-
function runNameFor(scenarioDir, harness) {
|
|
43
|
+
function runNameFor(scenarioDir, harness, control = false) {
|
|
41
44
|
const m = path.resolve(scenarioDir).match(/skills\/([^/]+)\/evals\/([^/]+)$/);
|
|
42
45
|
if (!m) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
|
|
43
46
|
const name = `${encodeRunNamePart(m[1])}--${encodeRunNamePart(m[2])}`;
|
|
44
|
-
const full = harness === "claude" ? name : `${name}--${harness}`;
|
|
47
|
+
const full = `${harness === "claude" ? name : `${name}--${harness}`}${control ? "--control" : ""}`;
|
|
45
48
|
if (Buffer.byteLength(`${full}.json.attempt`) <= 255) return full;
|
|
46
|
-
return `~v3~${createHash("sha256").update(JSON.stringify([
|
|
49
|
+
return `~v3~${createHash("sha256").update(JSON.stringify(control ? [
|
|
50
|
+
m[1],
|
|
51
|
+
m[2],
|
|
52
|
+
harness,
|
|
53
|
+
"control"
|
|
54
|
+
] : [
|
|
47
55
|
m[1],
|
|
48
56
|
m[2],
|
|
49
57
|
harness
|
|
@@ -71,7 +79,7 @@ const RESERVED = /* @__PURE__ */ new Set([
|
|
|
71
79
|
".grok",
|
|
72
80
|
"node_modules"
|
|
73
81
|
]);
|
|
74
|
-
function materialize(s, runDir, harness) {
|
|
82
|
+
function materialize(s, runDir, harness, control = false) {
|
|
75
83
|
const workdir = path.join(runDir, "workdir");
|
|
76
84
|
fs.rmSync(runDir, {
|
|
77
85
|
recursive: true,
|
|
@@ -95,7 +103,7 @@ function materialize(s, runDir, harness) {
|
|
|
95
103
|
fs.mkdirSync(path.dirname(dest), { recursive: true });
|
|
96
104
|
fs.writeFileSync(dest, content);
|
|
97
105
|
}
|
|
98
|
-
const roots = harness === "codex" ? [".claude", ".agents"] : harness === "grok" ? [".grok"] : [".claude"];
|
|
106
|
+
const roots = control ? [] : harness === "codex" ? [".claude", ".agents"] : harness === "grok" ? [".grok"] : [".claude"];
|
|
99
107
|
for (const root of roots) fs.cpSync(s.skillDir, path.join(workdir, root, "skills", s.skill), {
|
|
100
108
|
recursive: true,
|
|
101
109
|
filter: (src) => path.basename(src) !== "evals"
|
|
@@ -140,7 +148,9 @@ function agentProvider(opts, workdir, skill, paths) {
|
|
|
140
148
|
skip_git_repo_check: true,
|
|
141
149
|
enable_streaming: true,
|
|
142
150
|
sandbox_mode: "workspace-write",
|
|
143
|
-
|
|
151
|
+
network_access_enabled: true,
|
|
152
|
+
web_search_enabled: true,
|
|
153
|
+
cli_env: { CODEX_HOME: path.join(workdir, "..", "..", "codex-home") }
|
|
144
154
|
}
|
|
145
155
|
};
|
|
146
156
|
return {
|
|
@@ -151,14 +161,17 @@ function agentProvider(opts, workdir, skill, paths) {
|
|
|
151
161
|
apiKeyRequired: false,
|
|
152
162
|
working_dir: workdir,
|
|
153
163
|
setting_sources: ["project"],
|
|
154
|
-
skills: [skill],
|
|
164
|
+
...opts.control ? {} : { skills: [skill] },
|
|
155
165
|
permission_mode: "acceptEdits",
|
|
156
166
|
append_allowed_tools: [
|
|
157
167
|
"Read",
|
|
158
168
|
"Write",
|
|
159
169
|
"Edit",
|
|
160
170
|
"Glob",
|
|
161
|
-
"Grep"
|
|
171
|
+
"Grep",
|
|
172
|
+
"Bash",
|
|
173
|
+
"WebFetch",
|
|
174
|
+
"WebSearch"
|
|
162
175
|
],
|
|
163
176
|
max_turns: opts.maxTurns ?? 50
|
|
164
177
|
}
|
|
@@ -218,7 +231,7 @@ function buildConfig(s, trials, opts, paths) {
|
|
|
218
231
|
description: s.criteria.context,
|
|
219
232
|
providers: [trialLabel(i)],
|
|
220
233
|
vars: {
|
|
221
|
-
task: s.prompt,
|
|
234
|
+
task: opts.control ? s.task : s.prompt,
|
|
222
235
|
workdir: t.workdir,
|
|
223
236
|
manifest: t.manifestPath
|
|
224
237
|
},
|
|
@@ -231,8 +244,13 @@ function buildConfig(s, trials, opts, paths) {
|
|
|
231
244
|
weight: item.max_score
|
|
232
245
|
}))
|
|
233
246
|
}, {
|
|
234
|
-
type: "
|
|
235
|
-
value:
|
|
247
|
+
type: "javascript",
|
|
248
|
+
value: `file://${paths.skillEvidencePath}`,
|
|
249
|
+
metric: "skill-used",
|
|
250
|
+
config: {
|
|
251
|
+
skill: s.skill,
|
|
252
|
+
required: !opts.control && s.criteria.skill_use !== "optional"
|
|
253
|
+
}
|
|
236
254
|
}]
|
|
237
255
|
}))
|
|
238
256
|
};
|
|
@@ -259,15 +277,31 @@ function requiredEvalPackages(opts, hasAnthropicKey) {
|
|
|
259
277
|
if (opts.harness === "codex") pkgs.push("@openai/codex-sdk");
|
|
260
278
|
return pkgs;
|
|
261
279
|
}
|
|
280
|
+
function privateCodexHome(dir) {
|
|
281
|
+
const source = process.env.CODEX_HOME ?? path.join(os.homedir(), ".codex");
|
|
282
|
+
fs.rmSync(dir, {
|
|
283
|
+
recursive: true,
|
|
284
|
+
force: true
|
|
285
|
+
});
|
|
286
|
+
fs.mkdirSync(dir, {
|
|
287
|
+
recursive: true,
|
|
288
|
+
mode: 448
|
|
289
|
+
});
|
|
290
|
+
for (const file of ["config.toml", "auth.json"]) {
|
|
291
|
+
const from = path.join(source, file);
|
|
292
|
+
if (fs.existsSync(from)) fs.symlinkSync(from, path.join(dir, file));
|
|
293
|
+
}
|
|
294
|
+
}
|
|
262
295
|
function generateRun(scenarioDir, opts, paths) {
|
|
263
296
|
const s = loadScenario(scenarioDir);
|
|
264
|
-
const name = runNameFor(scenarioDir, opts.harness);
|
|
297
|
+
const name = runNameFor(scenarioDir, opts.harness, opts.control);
|
|
265
298
|
const runDir = path.join(paths.scratchDir, name);
|
|
266
299
|
fs.rmSync(runDir, {
|
|
267
300
|
recursive: true,
|
|
268
301
|
force: true
|
|
269
302
|
});
|
|
270
|
-
const trials = Array.from({ length: opts.trials ?? 1 }, (_, i) => materialize(s, path.join(runDir, trialLabel(i)), opts.harness));
|
|
303
|
+
const trials = Array.from({ length: opts.trials ?? 1 }, (_, i) => materialize(s, path.join(runDir, trialLabel(i)), opts.harness, opts.control));
|
|
304
|
+
if (opts.harness === "codex") privateCodexHome(path.join(runDir, "codex-home"));
|
|
271
305
|
const sdkDir = sdkNodeModulesDir();
|
|
272
306
|
if (sdkDir !== void 0) {
|
|
273
307
|
const link = path.join(runDir, "node_modules");
|
|
@@ -288,4 +322,4 @@ function generateRun(scenarioDir, opts, paths) {
|
|
|
288
322
|
};
|
|
289
323
|
}
|
|
290
324
|
//#endregion
|
|
291
|
-
export { DEFAULT_CLAUDE_AGENT, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
|
|
325
|
+
export { DEFAULT_CLAUDE_AGENT, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, privateCodexHome, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
import fs from "node:fs";
|
|
2
|
+
import path from "node:path";
|
|
3
|
+
//#region src/skill-evidence.ts
|
|
4
|
+
const SKILL_ROOTS = [
|
|
5
|
+
".claude",
|
|
6
|
+
".agents",
|
|
7
|
+
".grok"
|
|
8
|
+
];
|
|
9
|
+
function real(file) {
|
|
10
|
+
try {
|
|
11
|
+
return fs.realpathSync(file);
|
|
12
|
+
} catch {
|
|
13
|
+
return;
|
|
14
|
+
}
|
|
15
|
+
}
|
|
16
|
+
function skillEvidence(context) {
|
|
17
|
+
const skill = context.config?.skill;
|
|
18
|
+
const workdir = context.vars.workdir;
|
|
19
|
+
if (typeof skill !== "string" || typeof workdir !== "string") return void 0;
|
|
20
|
+
const calls = (value) => Array.isArray(value) ? value.filter((c) => c !== null && typeof c === "object") : [];
|
|
21
|
+
const metadata = context.metadata ?? context.providerResponse?.metadata;
|
|
22
|
+
if (calls(metadata?.skillCalls).find((c) => c.name === skill && c.is_error !== true)) return `skill call ${skill}`;
|
|
23
|
+
const installed = new Set(SKILL_ROOTS.map((root) => real(path.join(workdir, root, "skills", skill, "SKILL.md"))).filter((p) => p !== void 0));
|
|
24
|
+
const root = real(workdir) ?? workdir;
|
|
25
|
+
if (calls(metadata?.toolCalls).find((c) => {
|
|
26
|
+
if (c.name !== "Read" || c.is_error !== false || typeof c.output !== "string") return false;
|
|
27
|
+
const file = c.input?.file_path;
|
|
28
|
+
if (typeof file !== "string") return false;
|
|
29
|
+
const named = path.resolve(workdir, file);
|
|
30
|
+
const inside = [workdir, root].some((w) => {
|
|
31
|
+
const rel = path.relative(w, named);
|
|
32
|
+
return rel !== ".." && !rel.startsWith(`..${path.sep}`) && !path.isAbsolute(rel);
|
|
33
|
+
});
|
|
34
|
+
const target = real(named);
|
|
35
|
+
return inside && target !== void 0 && installed.has(target);
|
|
36
|
+
})) return `read of the installed ${skill}/SKILL.md`;
|
|
37
|
+
}
|
|
38
|
+
function assertSkillUsed(_output, context) {
|
|
39
|
+
const evidence = skillEvidence(context);
|
|
40
|
+
const required = context.config?.required !== false;
|
|
41
|
+
const used = evidence !== void 0;
|
|
42
|
+
return {
|
|
43
|
+
pass: used || !required,
|
|
44
|
+
score: used ? 1 : 0,
|
|
45
|
+
reason: used ? `skill used: ${evidence}` : `skill ${String(context.config?.skill)} not loaded${required ? "" : " (optional for this scenario)"}`
|
|
46
|
+
};
|
|
47
|
+
}
|
|
48
|
+
//#endregion
|
|
49
|
+
export { assertSkillUsed as default, skillEvidence };
|
package/dist/transform.js
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
1
2
|
import fs from "node:fs";
|
|
2
3
|
import path from "node:path";
|
|
3
|
-
import { createHash } from "node:crypto";
|
|
4
4
|
//#region src/transform.ts
|
|
5
|
-
const PER_FILE_CAP =
|
|
6
|
-
const TOTAL_CAP =
|
|
5
|
+
const PER_FILE_CAP = 16e3;
|
|
6
|
+
const TOTAL_CAP = 64e3;
|
|
7
7
|
function safeSlice(text, end) {
|
|
8
8
|
let cut = text.slice(0, end);
|
|
9
9
|
const last = cut.charCodeAt(cut.length - 1);
|
package/docs/adoption.md
CHANGED
package/docs/scenarios.md
CHANGED
|
@@ -54,7 +54,16 @@ item needs a non-empty `name` and `description` and a positive `max_score`.
|
|
|
54
54
|
Each item becomes one `llm-rubric` assertion weighted by `max_score`, inside an
|
|
55
55
|
assert-set with threshold 0.7. A separate `skill-used` assertion sits outside
|
|
56
56
|
that aggregate, so a run that produces good output without ever loading the
|
|
57
|
-
skill still fails.
|
|
57
|
+
skill still fails. The skill counts as loaded when the harness reports a skill
|
|
58
|
+
call for it, or when the agent completed a read of the installed
|
|
59
|
+
`<config-root>/skills/<skill>/SKILL.md` in its workdir. Claude Code often reads
|
|
60
|
+
the file directly instead of calling its Skill tool, and either way the same
|
|
61
|
+
instructions reach its context.
|
|
62
|
+
|
|
63
|
+
An out-of-lane scenario, where the right answer is to decline the skill, sets
|
|
64
|
+
`"skill_use": "optional"` in `criteria.json`. The assertion then always passes,
|
|
65
|
+
and the result still records whether the skill was loaded. Omitted, it is
|
|
66
|
+
`"required"`. There is no test-level threshold: both must pass. The
|
|
58
67
|
reported score is the assert-set's weighted score; skill-used is reported
|
|
59
68
|
separately as a rate across trials.
|
|
60
69
|
|
|
@@ -62,14 +71,28 @@ Write descriptions a judge can check against the deliverable: an observable
|
|
|
62
71
|
property, not a feeling. Weight the items that would make a reviewer reject the
|
|
63
72
|
work.
|
|
64
73
|
|
|
74
|
+
The agent runs online, like a real session: it has a shell and web access on
|
|
75
|
+
every harness (Claude: Bash, WebFetch, WebSearch; Codex: network and web search
|
|
76
|
+
in its `workspace-write` sandbox; Grok: its CLI defaults). It runs on the
|
|
77
|
+
operator's machine with the operator's logins (Codex gets a per-run
|
|
78
|
+
`CODEX_HOME` carrying only its config and login, so the operator's own skills
|
|
79
|
+
and global guidance stay out), so a task must never ask for a
|
|
80
|
+
live mutation such as posting a comment, pushing, publishing, or writing to a
|
|
81
|
+
shared workspace. Put that state in fixture files and grade the plan. Checks
|
|
82
|
+
the agent can run for itself (install, build, test, fetch a public page) are
|
|
83
|
+
fair to require.
|
|
84
|
+
|
|
85
|
+
Name a specific tool or version only when the skill teaches it. Otherwise grade
|
|
86
|
+
the property the tool provides, so an equivalent approach passes.
|
|
87
|
+
|
|
65
88
|
## What the judge sees
|
|
66
89
|
|
|
67
90
|
The agent's final message, plus every file in the workdir that differs from the
|
|
68
91
|
pre-run manifest. Unchanged inputs are omitted; deleted inputs, unreadable
|
|
69
92
|
files, and non-regular files are named rather than read.
|
|
70
93
|
|
|
71
|
-
Sections are sorted by path, each file is capped at
|
|
72
|
-
appended total at
|
|
94
|
+
Sections are sorted by path, each file is capped at 16,000 characters and the
|
|
95
|
+
appended total at 64,000, with truncation stated inline. Very large outputs make
|
|
73
96
|
rubric judges return nothing at all, which is why the caps exist. Keep fixtures
|
|
74
97
|
small enough that the deliverable fits.
|
|
75
98
|
|
|
@@ -84,8 +107,10 @@ frontmatter block; body text mentioning the key does not count.
|
|
|
84
107
|
|
|
85
108
|
## The workdir
|
|
86
109
|
|
|
87
|
-
Per run, under
|
|
88
|
-
|
|
110
|
+
Per run, under a scratch directory in the system temp dir
|
|
111
|
+
(`skillcheck-<hash of the root>/<name>/trial-<n>/`), rebuilt from scratch each
|
|
112
|
+
time. It stays outside the root so the agent cannot reach the skill's source,
|
|
113
|
+
its evals, or the repository's own agent guidance. The skill under test is installed where the harness discovers skills:
|
|
89
114
|
`.claude/skills/<skill>/`, plus `.agents/skills/<skill>/` on codex, or
|
|
90
115
|
`.grok/skills/<skill>/` on Grok, with its `evals/` directory excluded, so
|
|
91
116
|
criteria never leak into the agent's context.
|
package/docs/usage.md
CHANGED
|
@@ -38,7 +38,7 @@ skillcheck run <scenario-dir> --harness grok
|
|
|
38
38
|
skillcheck run <scenario-dir> --trials 3 --agent-effort medium
|
|
39
39
|
```
|
|
40
40
|
|
|
41
|
-
Materializes the scenario into
|
|
41
|
+
Materializes the scenario into a temp-dir workdir per trial ([workdir](scenarios.md#the-workdir)),
|
|
42
42
|
installs the skill under test into that workdir, drives the agent, and grades
|
|
43
43
|
the files it wrote. Exit 0 means pass, 1 means graded fail, and 2 means error.
|
|
44
44
|
Exit 2 covers missing usable promptfoo output or optional eval peers. The
|
|
@@ -82,6 +82,18 @@ So is a result with fewer rows than trials, or a row without its checklist and
|
|
|
82
82
|
With `--trials` above 1, `run` and `sweep` print min, spread, pass count,
|
|
83
83
|
skill-used count, and `NOISY`.
|
|
84
84
|
|
|
85
|
+
### Control
|
|
86
|
+
|
|
87
|
+
`--control` runs a scenario without the skill: nothing is installed in the
|
|
88
|
+
workdir, a hidden skill's explicit invocation is dropped from the task, and the
|
|
89
|
+
`skill-used` assertion never fails. Its result sits beside the skill run
|
|
90
|
+
(`<name>--control.json`) with `variant: "control"` in the sidecar, and `sweep
|
|
91
|
+
--control` covers every scenario. `summarize` pairs each scenario with its
|
|
92
|
+
control and adds two columns per skill: the mean control score, and the lift
|
|
93
|
+
(skill score minus control score over the paired scenarios). A scenario whose
|
|
94
|
+
control passes every trial prints as `NO LIFT`: it passes without the skill, so
|
|
95
|
+
it does not test the skill.
|
|
96
|
+
|
|
85
97
|
### Agent effort
|
|
86
98
|
|
|
87
99
|
`--agent-effort low|medium|high|xhigh|max` sets the Claude agent's effort.
|
|
@@ -96,8 +108,8 @@ Omitting it leaves Claude Code's default. Only `--harness claude` takes it;
|
|
|
96
108
|
workdir with the skill under `.grok/skills/`. It uses native streaming events
|
|
97
109
|
to count a completed read of that skill's `SKILL.md` as `skill-used` evidence.
|
|
98
110
|
Grok must be logged in locally or have its supported credentials configured.
|
|
99
|
-
The run disables
|
|
100
|
-
|
|
111
|
+
The run disables subagents and grants edit permission in the workdir; web
|
|
112
|
+
search stays on. `--agent` selects a Grok model ID.
|
|
101
113
|
|
|
102
114
|
`--judge` takes either a bare Claude model (graded through the Anthropic
|
|
103
115
|
selection in [auth](#auth)) or a provider-qualified promptfoo id, passed
|
|
@@ -200,6 +212,7 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
|
|
|
200
212
|
"judge_model": "claude-opus-5",
|
|
201
213
|
"judge_effort": null,
|
|
202
214
|
"trials": 3,
|
|
215
|
+
"agent_access": "online",
|
|
203
216
|
"aggregate": { "pass": false, "pass_rate": 0.6667, "score": 0.81, "...": "..." },
|
|
204
217
|
"ran_at": "<ISO timestamp>",
|
|
205
218
|
"tool_version": "<skillcheck version>"
|
|
@@ -207,7 +220,8 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
|
|
|
207
220
|
```
|
|
208
221
|
|
|
209
222
|
`summarize` reads those sidecars and refuses to mix skills-tree revisions or
|
|
210
|
-
run configurations (agent model and effort, judge model and effort, trials
|
|
223
|
+
run configurations (agent model and effort, judge model and effort, trials, and
|
|
224
|
+
agent access) in
|
|
211
225
|
one scorecard, including retained rows from partial reruns, unless
|
|
212
226
|
`--allow-mixed`. Configurations are compared within a harness, since harnesses
|
|
213
227
|
differ by design. Sidecars written before run configurations were recorded fall
|
|
@@ -218,9 +232,10 @@ becomes `mixed` and per-entry shas remain. A result with no sidecar reduces as
|
|
|
218
232
|
|
|
219
233
|
## State
|
|
220
234
|
|
|
221
|
-
`<root>/.skillcheck/` holds `
|
|
222
|
-
|
|
223
|
-
written inside the installed
|
|
235
|
+
`<root>/.skillcheck/` holds `results/`, disposable and safe to gitignore, and
|
|
236
|
+
`scorecards/`, which is meant to be committed. Scratch workdirs live under the
|
|
237
|
+
system temp dir, outside the root. Nothing is ever written inside the installed
|
|
238
|
+
package.
|
|
224
239
|
|
|
225
240
|
## Auth
|
|
226
241
|
|