@uinaf/skillcheck 1.1.0 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/cli.js +28 -11
- package/dist/scenario.js +15 -5
- package/dist/skill-evidence.js +49 -0
- package/dist/transform.js +3 -3
- package/docs/adoption.md +0 -1
- package/docs/scenarios.md +29 -5
- package/docs/usage.md +15 -7
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -45,7 +45,7 @@ current directory.
|
|
|
45
45
|
<root>/skills/<skill>/SKILL.md linted
|
|
46
46
|
<root>/skills/<skill>/evals/<scenario>/task.md the problem + input files
|
|
47
47
|
<root>/skills/<skill>/evals/<scenario>/criteria.json the weighted checklist
|
|
48
|
-
<root>/.skillcheck/ results,
|
|
48
|
+
<root>/.skillcheck/ results, scorecards
|
|
49
49
|
```
|
|
50
50
|
|
|
51
51
|
`cli/*/skills/<skill>/` is scanned too, for repos that keep a skill next to the
|
package/dist/cli.js
CHANGED
|
@@ -2,7 +2,9 @@
|
|
|
2
2
|
import { lintSkills } from "./lint.js";
|
|
3
3
|
import { encodeRunNamePart, generateRun, requiredEvalPackages, resolvePackageDir, runNameFor } from "./scenario.js";
|
|
4
4
|
import { execFileSync, spawnSync } from "node:child_process";
|
|
5
|
+
import { createHash } from "node:crypto";
|
|
5
6
|
import fs from "node:fs";
|
|
7
|
+
import os from "node:os";
|
|
6
8
|
import path from "node:path";
|
|
7
9
|
import { fileURLToPath } from "node:url";
|
|
8
10
|
//#region src/cli.ts
|
|
@@ -61,9 +63,10 @@ function resolveRoot(flags) {
|
|
|
61
63
|
}
|
|
62
64
|
function stateDirs(root) {
|
|
63
65
|
const base = path.join(root, ".skillcheck");
|
|
66
|
+
const id = createHash("sha256").update(path.resolve(root)).digest("hex").slice(0, 12);
|
|
64
67
|
return {
|
|
65
68
|
results: path.join(base, "results"),
|
|
66
|
-
scratch: path.join(
|
|
69
|
+
scratch: path.join(fs.realpathSync(os.tmpdir()), `skillcheck-${id}`),
|
|
67
70
|
scorecards: path.join(base, "scorecards")
|
|
68
71
|
};
|
|
69
72
|
}
|
|
@@ -80,13 +83,13 @@ function runOptions(flags) {
|
|
|
80
83
|
const judgeModel = flags.get("--judge") ?? "claude-opus-5";
|
|
81
84
|
const judgeEffort = flags.get("--judge-effort");
|
|
82
85
|
if (judgeEffort !== void 0) {
|
|
83
|
-
|
|
86
|
+
const levels = judgeModel.includes(":") ? [
|
|
84
87
|
"minimal",
|
|
85
88
|
"low",
|
|
86
89
|
"medium",
|
|
87
90
|
"high"
|
|
88
|
-
]
|
|
89
|
-
if (!
|
|
91
|
+
] : AGENT_EFFORTS;
|
|
92
|
+
if (!levels.includes(judgeEffort)) throw new Error(`--judge-effort for ${judgeModel} must be ${levels.join(", ")}, got ${judgeEffort}`);
|
|
90
93
|
}
|
|
91
94
|
return {
|
|
92
95
|
harness,
|
|
@@ -180,12 +183,12 @@ function classifyRow(raw, stats) {
|
|
|
180
183
|
if (typeof res.score !== "number" || typeof res.success !== "boolean") return { error: "promptfoo result carried no usable score" };
|
|
181
184
|
const components = Array.isArray(res.gradingResult?.componentResults) ? res.gradingResult.componentResults : [];
|
|
182
185
|
const checklist = components.find((c) => c?.metadata?.assertionSet?.type === "assert-set");
|
|
183
|
-
const skillUsed = components.find((c) => c?.assertion?.type === "skill-used");
|
|
184
|
-
if (typeof checklist?.score !== "number" || typeof skillUsed?.
|
|
186
|
+
const skillUsed = components.find((c) => c?.assertion?.type === "skill-used" || c?.assertion?.metric === "skill-used");
|
|
187
|
+
if (typeof checklist?.score !== "number" || typeof skillUsed?.score !== "number") return { error: "promptfoo result carried no checklist or skill-used verdict" };
|
|
185
188
|
return {
|
|
186
189
|
score: checklist.score,
|
|
187
190
|
pass: res.success,
|
|
188
|
-
skillUsed: skillUsed.
|
|
191
|
+
skillUsed: skillUsed.score >= 1
|
|
189
192
|
};
|
|
190
193
|
}
|
|
191
194
|
const NOISY_SPREAD = .2;
|
|
@@ -235,12 +238,24 @@ function metaPath(resultPath) {
|
|
|
235
238
|
function attemptPath(resultPath) {
|
|
236
239
|
return `${resultPath}.attempt`;
|
|
237
240
|
}
|
|
241
|
+
function ensurePrivateDir(dir) {
|
|
242
|
+
fs.mkdirSync(dir, {
|
|
243
|
+
recursive: true,
|
|
244
|
+
mode: 448
|
|
245
|
+
});
|
|
246
|
+
const st = fs.lstatSync(dir);
|
|
247
|
+
const uid = process.getuid?.();
|
|
248
|
+
if (!st.isDirectory() || uid !== void 0 && st.uid !== uid) throw new Error(`scratch dir ${dir} is not a directory owned by this user`);
|
|
249
|
+
if ((st.mode & 63) !== 0) fs.chmodSync(dir, 448);
|
|
250
|
+
}
|
|
238
251
|
function runScenario(scenarioDir, opts, root) {
|
|
239
252
|
const dirs = stateDirs(root);
|
|
253
|
+
ensurePrivateDir(dirs.scratch);
|
|
240
254
|
const { name, configPath, skill, scenario } = generateRun(path.resolve(scenarioDir), opts, {
|
|
241
255
|
scratchDir: dirs.scratch,
|
|
242
256
|
transformPath: path.join(here, `transform${selfExt}`),
|
|
243
|
-
grokProviderPath: path.join(here, `grok-provider${selfExt}`)
|
|
257
|
+
grokProviderPath: path.join(here, `grok-provider${selfExt}`),
|
|
258
|
+
skillEvidencePath: path.join(here, `skill-evidence${selfExt}`)
|
|
244
259
|
});
|
|
245
260
|
fs.mkdirSync(dirs.results, { recursive: true });
|
|
246
261
|
const resultPath = path.join(dirs.results, `${name}.json`);
|
|
@@ -304,7 +319,8 @@ function judgeName(judge) {
|
|
|
304
319
|
if (typeof judge === "string") return judge.replace(/^anthropic:messages:/, "");
|
|
305
320
|
const j = judge;
|
|
306
321
|
const name = j?.config?.model ?? j?.id;
|
|
307
|
-
|
|
322
|
+
if (typeof name !== "string") return "unknown";
|
|
323
|
+
return j?.config?.effort === void 0 ? name : name.replace(/^anthropic:messages:/, "");
|
|
308
324
|
}
|
|
309
325
|
function resultRunConfig(raw, meta, harness) {
|
|
310
326
|
const m = meta;
|
|
@@ -318,7 +334,8 @@ function resultRunConfig(raw, meta, harness) {
|
|
|
318
334
|
const r = raw;
|
|
319
335
|
const agent = r?.config?.providers?.[0]?.config;
|
|
320
336
|
const judge = r?.config?.defaultTest?.options?.provider;
|
|
321
|
-
const
|
|
337
|
+
const judgeConfig = judge?.config;
|
|
338
|
+
const judgeEffort = judgeConfig?.reasoning_effort ?? judgeConfig?.effort;
|
|
322
339
|
return {
|
|
323
340
|
agent_model: typeof agent?.model === "string" ? agent.model : `${harness}-default`,
|
|
324
341
|
agent_effort: typeof agent?.effort === "string" ? agent.effort : null,
|
|
@@ -633,4 +650,4 @@ if (isMainModule()) {
|
|
|
633
650
|
}
|
|
634
651
|
}
|
|
635
652
|
//#endregion
|
|
636
|
-
export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, formatStats, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
|
|
653
|
+
export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, ensurePrivateDir, formatStats, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
|
package/dist/scenario.js
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { createRequire } from "node:module";
|
|
2
|
-
import fs from "node:fs";
|
|
3
|
-
import path from "node:path";
|
|
4
2
|
import { createHash } from "node:crypto";
|
|
3
|
+
import fs from "node:fs";
|
|
5
4
|
import os from "node:os";
|
|
5
|
+
import path from "node:path";
|
|
6
6
|
//#region src/scenario.ts
|
|
7
7
|
const DEFAULT_CLAUDE_AGENT = "claude-opus-5";
|
|
8
8
|
function loadScenario(scenarioDir) {
|
|
@@ -12,6 +12,7 @@ function loadScenario(scenarioDir) {
|
|
|
12
12
|
const taskMd = fs.readFileSync(path.join(scenarioDir, "task.md"), "utf8");
|
|
13
13
|
const criteria = JSON.parse(fs.readFileSync(path.join(scenarioDir, "criteria.json"), "utf8"));
|
|
14
14
|
if (criteria.type !== "weighted_checklist" || !Array.isArray(criteria.checklist) || criteria.checklist.length === 0) throw new Error(`unsupported or empty criteria in ${scenarioDir}`);
|
|
15
|
+
if (criteria.skill_use !== void 0 && criteria.skill_use !== "required" && criteria.skill_use !== "optional") throw new Error(`skill_use must be "required" or "optional" in ${scenarioDir}`);
|
|
15
16
|
for (const item of criteria.checklist) if (!(typeof item?.name === "string" && item.name.trim() !== "" && typeof item?.description === "string" && item.description.trim() !== "" && Number.isFinite(item?.max_score) && item.max_score > 0)) throw new Error(`invalid checklist item in ${scenarioDir}: ${JSON.stringify(item)}`);
|
|
16
17
|
const files = [];
|
|
17
18
|
let prompt = taskMd.replace(/^=+ FILE: (.+?) =+\n([\s\S]*?)\n=+ END FILE =+$/gm, (_, name, content) => {
|
|
@@ -179,10 +180,14 @@ function buildConfig(s, trials, opts, paths) {
|
|
|
179
180
|
provider: opts.judgeModel.includes(":") ? opts.judgeEffort === void 0 ? opts.judgeModel : {
|
|
180
181
|
id: opts.judgeModel,
|
|
181
182
|
config: { reasoning_effort: opts.judgeEffort }
|
|
182
|
-
} : process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
|
|
183
|
+
} : process.env.ANTHROPIC_API_KEY ? opts.judgeEffort === void 0 ? `anthropic:messages:${opts.judgeModel}` : {
|
|
184
|
+
id: `anthropic:messages:${opts.judgeModel}`,
|
|
185
|
+
config: { effort: opts.judgeEffort }
|
|
186
|
+
} : {
|
|
183
187
|
id: "anthropic:claude-agent-sdk",
|
|
184
188
|
config: {
|
|
185
189
|
model: opts.judgeModel,
|
|
190
|
+
...opts.judgeEffort ? { effort: opts.judgeEffort } : {},
|
|
186
191
|
apiKeyRequired: false,
|
|
187
192
|
max_turns: 3,
|
|
188
193
|
output_format: {
|
|
@@ -227,8 +232,13 @@ function buildConfig(s, trials, opts, paths) {
|
|
|
227
232
|
weight: item.max_score
|
|
228
233
|
}))
|
|
229
234
|
}, {
|
|
230
|
-
type: "
|
|
231
|
-
value:
|
|
235
|
+
type: "javascript",
|
|
236
|
+
value: `file://${paths.skillEvidencePath}`,
|
|
237
|
+
metric: "skill-used",
|
|
238
|
+
config: {
|
|
239
|
+
skill: s.skill,
|
|
240
|
+
required: s.criteria.skill_use !== "optional"
|
|
241
|
+
}
|
|
232
242
|
}]
|
|
233
243
|
}))
|
|
234
244
|
};
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
import fs from "node:fs";
|
|
2
|
+
import path from "node:path";
|
|
3
|
+
//#region src/skill-evidence.ts
|
|
4
|
+
const SKILL_ROOTS = [
|
|
5
|
+
".claude",
|
|
6
|
+
".agents",
|
|
7
|
+
".grok"
|
|
8
|
+
];
|
|
9
|
+
function real(file) {
|
|
10
|
+
try {
|
|
11
|
+
return fs.realpathSync(file);
|
|
12
|
+
} catch {
|
|
13
|
+
return;
|
|
14
|
+
}
|
|
15
|
+
}
|
|
16
|
+
function skillEvidence(context) {
|
|
17
|
+
const skill = context.config?.skill;
|
|
18
|
+
const workdir = context.vars.workdir;
|
|
19
|
+
if (typeof skill !== "string" || typeof workdir !== "string") return void 0;
|
|
20
|
+
const calls = (value) => Array.isArray(value) ? value.filter((c) => c !== null && typeof c === "object") : [];
|
|
21
|
+
const metadata = context.metadata ?? context.providerResponse?.metadata;
|
|
22
|
+
if (calls(metadata?.skillCalls).find((c) => c.name === skill && c.is_error !== true)) return `skill call ${skill}`;
|
|
23
|
+
const installed = new Set(SKILL_ROOTS.map((root) => real(path.join(workdir, root, "skills", skill, "SKILL.md"))).filter((p) => p !== void 0));
|
|
24
|
+
const root = real(workdir) ?? workdir;
|
|
25
|
+
if (calls(metadata?.toolCalls).find((c) => {
|
|
26
|
+
if (c.name !== "Read" || c.is_error !== false || typeof c.output !== "string") return false;
|
|
27
|
+
const file = c.input?.file_path;
|
|
28
|
+
if (typeof file !== "string") return false;
|
|
29
|
+
const named = path.resolve(workdir, file);
|
|
30
|
+
const inside = [workdir, root].some((w) => {
|
|
31
|
+
const rel = path.relative(w, named);
|
|
32
|
+
return rel !== ".." && !rel.startsWith(`..${path.sep}`) && !path.isAbsolute(rel);
|
|
33
|
+
});
|
|
34
|
+
const target = real(named);
|
|
35
|
+
return inside && target !== void 0 && installed.has(target);
|
|
36
|
+
})) return `read of the installed ${skill}/SKILL.md`;
|
|
37
|
+
}
|
|
38
|
+
function assertSkillUsed(_output, context) {
|
|
39
|
+
const evidence = skillEvidence(context);
|
|
40
|
+
const required = context.config?.required !== false;
|
|
41
|
+
const used = evidence !== void 0;
|
|
42
|
+
return {
|
|
43
|
+
pass: used || !required,
|
|
44
|
+
score: used ? 1 : 0,
|
|
45
|
+
reason: used ? `skill used: ${evidence}` : `skill ${String(context.config?.skill)} not loaded${required ? "" : " (optional for this scenario)"}`
|
|
46
|
+
};
|
|
47
|
+
}
|
|
48
|
+
//#endregion
|
|
49
|
+
export { assertSkillUsed as default, skillEvidence };
|
package/dist/transform.js
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
1
2
|
import fs from "node:fs";
|
|
2
3
|
import path from "node:path";
|
|
3
|
-
import { createHash } from "node:crypto";
|
|
4
4
|
//#region src/transform.ts
|
|
5
|
-
const PER_FILE_CAP =
|
|
6
|
-
const TOTAL_CAP =
|
|
5
|
+
const PER_FILE_CAP = 16e3;
|
|
6
|
+
const TOTAL_CAP = 64e3;
|
|
7
7
|
function safeSlice(text, end) {
|
|
8
8
|
let cut = text.slice(0, end);
|
|
9
9
|
const last = cut.charCodeAt(cut.length - 1);
|
package/docs/adoption.md
CHANGED
package/docs/scenarios.md
CHANGED
|
@@ -54,7 +54,16 @@ item needs a non-empty `name` and `description` and a positive `max_score`.
|
|
|
54
54
|
Each item becomes one `llm-rubric` assertion weighted by `max_score`, inside an
|
|
55
55
|
assert-set with threshold 0.7. A separate `skill-used` assertion sits outside
|
|
56
56
|
that aggregate, so a run that produces good output without ever loading the
|
|
57
|
-
skill still fails.
|
|
57
|
+
skill still fails. The skill counts as loaded when the harness reports a skill
|
|
58
|
+
call for it, or when the agent completed a read of the installed
|
|
59
|
+
`<config-root>/skills/<skill>/SKILL.md` in its workdir. Claude Code often reads
|
|
60
|
+
the file directly instead of calling its Skill tool, and either way the same
|
|
61
|
+
instructions reach its context.
|
|
62
|
+
|
|
63
|
+
An out-of-lane scenario, where the right answer is to decline the skill, sets
|
|
64
|
+
`"skill_use": "optional"` in `criteria.json`. The assertion then always passes,
|
|
65
|
+
and the result still records whether the skill was loaded. Omitted, it is
|
|
66
|
+
`"required"`. There is no test-level threshold: both must pass. The
|
|
58
67
|
reported score is the assert-set's weighted score; skill-used is reported
|
|
59
68
|
separately as a rate across trials.
|
|
60
69
|
|
|
@@ -62,14 +71,27 @@ Write descriptions a judge can check against the deliverable: an observable
|
|
|
62
71
|
property, not a feeling. Weight the items that would make a reviewer reject the
|
|
63
72
|
work.
|
|
64
73
|
|
|
74
|
+
On the Claude harness the agent can read, search, and write files in its
|
|
75
|
+
workdir, but has no shell, so it cannot install, build, test, or reach the
|
|
76
|
+
network. The shell stays off because the agent runs on the operator's machine
|
|
77
|
+
with the operator's credentials. Codex runs shell commands inside its
|
|
78
|
+
`workspace-write` sandbox, and Grok follows its own CLI permission mode. Write criteria a shell-less agent can meet, so one scenario
|
|
79
|
+
scores comparably across harnesses: a checklist item that requires live proof,
|
|
80
|
+
such as a frozen lockfile or a verified release, fails on Claude. Grade whether
|
|
81
|
+
the deliverable names the checks it could not run and hands them over
|
|
82
|
+
precisely.
|
|
83
|
+
|
|
84
|
+
Name a specific tool or version only when the skill teaches it. Otherwise grade
|
|
85
|
+
the property the tool provides, so an equivalent approach passes.
|
|
86
|
+
|
|
65
87
|
## What the judge sees
|
|
66
88
|
|
|
67
89
|
The agent's final message, plus every file in the workdir that differs from the
|
|
68
90
|
pre-run manifest. Unchanged inputs are omitted; deleted inputs, unreadable
|
|
69
91
|
files, and non-regular files are named rather than read.
|
|
70
92
|
|
|
71
|
-
Sections are sorted by path, each file is capped at
|
|
72
|
-
appended total at
|
|
93
|
+
Sections are sorted by path, each file is capped at 16,000 characters and the
|
|
94
|
+
appended total at 64,000, with truncation stated inline. Very large outputs make
|
|
73
95
|
rubric judges return nothing at all, which is why the caps exist. Keep fixtures
|
|
74
96
|
small enough that the deliverable fits.
|
|
75
97
|
|
|
@@ -84,8 +106,10 @@ frontmatter block; body text mentioning the key does not count.
|
|
|
84
106
|
|
|
85
107
|
## The workdir
|
|
86
108
|
|
|
87
|
-
Per run, under
|
|
88
|
-
|
|
109
|
+
Per run, under a scratch directory in the system temp dir
|
|
110
|
+
(`skillcheck-<hash of the root>/<name>/trial-<n>/`), rebuilt from scratch each
|
|
111
|
+
time. It stays outside the root so the agent cannot reach the skill's source,
|
|
112
|
+
its evals, or the repository's own agent guidance. The skill under test is installed where the harness discovers skills:
|
|
89
113
|
`.claude/skills/<skill>/`, plus `.agents/skills/<skill>/` on codex, or
|
|
90
114
|
`.grok/skills/<skill>/` on Grok, with its `evals/` directory excluded, so
|
|
91
115
|
criteria never leak into the agent's context.
|
package/docs/usage.md
CHANGED
|
@@ -38,7 +38,7 @@ skillcheck run <scenario-dir> --harness grok
|
|
|
38
38
|
skillcheck run <scenario-dir> --trials 3 --agent-effort medium
|
|
39
39
|
```
|
|
40
40
|
|
|
41
|
-
Materializes the scenario into
|
|
41
|
+
Materializes the scenario into a temp-dir workdir per trial ([workdir](scenarios.md#the-workdir)),
|
|
42
42
|
installs the skill under test into that workdir, drives the agent, and grades
|
|
43
43
|
the files it wrote. Exit 0 means pass, 1 means graded fail, and 2 means error.
|
|
44
44
|
Exit 2 covers missing usable promptfoo output or optional eval peers. The
|
|
@@ -109,9 +109,16 @@ skillcheck run <scenario-dir> --judge openai:chat:gpt-5.6-sol --judge-effort hig
|
|
|
109
109
|
|
|
110
110
|
A provider-qualified judge authenticates through that provider's own env
|
|
111
111
|
(`OPENAI_API_KEY`, plus `OPENAI_BASE_URL` for a gateway) and is recorded
|
|
112
|
-
verbatim in the scorecard's `judge_model` column. `--judge-effort`
|
|
113
|
-
(minimal|low|medium|high) sets `reasoning_effort
|
|
114
|
-
|
|
112
|
+
verbatim in the scorecard's `judge_model` column. For it, `--judge-effort`
|
|
113
|
+
(minimal|low|medium|high) sets `reasoning_effort`. For a bare Claude judge,
|
|
114
|
+
`--judge-effort` takes Claude's levels (low|medium|high|xhigh|max) and is passed
|
|
115
|
+
as `effort` on either Anthropic path; the SDK judge starts Claude Code with
|
|
116
|
+
`--effort`:
|
|
117
|
+
|
|
118
|
+
```sh
|
|
119
|
+
skillcheck run <scenario-dir> --agent claude-opus-5-5 --agent-effort medium \
|
|
120
|
+
--judge claude-opus-5-5 --judge-effort high --trials 3
|
|
121
|
+
```
|
|
115
122
|
|
|
116
123
|
## Sweep
|
|
117
124
|
|
|
@@ -211,9 +218,10 @@ becomes `mixed` and per-entry shas remain. A result with no sidecar reduces as
|
|
|
211
218
|
|
|
212
219
|
## State
|
|
213
220
|
|
|
214
|
-
`<root>/.skillcheck/` holds `
|
|
215
|
-
|
|
216
|
-
written inside the installed
|
|
221
|
+
`<root>/.skillcheck/` holds `results/`, disposable and safe to gitignore, and
|
|
222
|
+
`scorecards/`, which is meant to be committed. Scratch workdirs live under the
|
|
223
|
+
system temp dir, outside the root. Nothing is ever written inside the installed
|
|
224
|
+
package.
|
|
217
225
|
|
|
218
226
|
## Auth
|
|
219
227
|
|