@uinaf/skillcheck 1.6.1 → 1.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/scenario.js +45 -9
- package/dist/skill-evidence.js +7 -5
- package/docs/adoption.md +3 -1
- package/docs/scenarios.md +21 -1
- package/docs/usage.md +13 -10
- package/package.json +1 -1
package/dist/scenario.js
CHANGED
|
@@ -14,7 +14,11 @@ function loadScenario(scenarioDir) {
|
|
|
14
14
|
const taskMd = fs.readFileSync(path.join(scenarioDir, "task.md"), "utf8");
|
|
15
15
|
const criteria = JSON.parse(fs.readFileSync(path.join(scenarioDir, "criteria.json"), "utf8"));
|
|
16
16
|
if (criteria.type !== "weighted_checklist" || !Array.isArray(criteria.checklist) || criteria.checklist.length === 0) throw new Error(`unsupported or empty criteria in ${scenarioDir}`);
|
|
17
|
-
if (criteria.skill_use !== void 0 &&
|
|
17
|
+
if (criteria.skill_use !== void 0 && ![
|
|
18
|
+
"required",
|
|
19
|
+
"optional",
|
|
20
|
+
"forbidden"
|
|
21
|
+
].includes(criteria.skill_use)) throw new Error(`skill_use must be "required", "optional", or "forbidden" in ${scenarioDir}`);
|
|
18
22
|
for (const item of criteria.checklist) if (!(typeof item?.name === "string" && item.name.trim() !== "" && typeof item?.description === "string" && item.description.trim() !== "" && Number.isFinite(item?.max_score) && item.max_score > 0)) throw new Error(`invalid checklist item in ${scenarioDir}: ${JSON.stringify(item)}`);
|
|
19
23
|
const files = [];
|
|
20
24
|
const task = taskMd.replace(FILE_BLOCK, (_, name, content) => {
|
|
@@ -26,18 +30,41 @@ function loadScenario(scenarioDir) {
|
|
|
26
30
|
});
|
|
27
31
|
let prompt = task;
|
|
28
32
|
const skillDir = path.resolve(scenarioDir, "../..");
|
|
29
|
-
if (isHiddenSkill(fs.readFileSync(path.join(skillDir, "SKILL.md"), "utf8")))
|
|
33
|
+
if (isHiddenSkill(fs.readFileSync(path.join(skillDir, "SKILL.md"), "utf8"))) {
|
|
34
|
+
if (criteria.skill_use === "forbidden" || criteria.install !== void 0) throw new Error(`a hidden skill is always invoked, so it has no routing to test: ${scenarioDir}`);
|
|
35
|
+
prompt = `Use the ${skill} skill for this task.\n\n${prompt}`;
|
|
36
|
+
}
|
|
37
|
+
const alternatives = resolveInstall(skillDir, criteria.install, scenarioDir);
|
|
30
38
|
return {
|
|
31
39
|
skill,
|
|
32
40
|
scenario,
|
|
33
41
|
name: `${skill}--${scenario}`,
|
|
34
42
|
skillDir,
|
|
43
|
+
alternatives,
|
|
35
44
|
prompt,
|
|
36
45
|
task,
|
|
37
46
|
files,
|
|
38
47
|
criteria
|
|
39
48
|
};
|
|
40
49
|
}
|
|
50
|
+
function resolveInstall(skillDir, install, scenarioDir) {
|
|
51
|
+
if (install === void 0) return [];
|
|
52
|
+
const skillsRoot = path.dirname(skillDir);
|
|
53
|
+
const self = path.basename(skillDir);
|
|
54
|
+
const invocable = (name) => {
|
|
55
|
+
const dir = path.join(skillsRoot, name);
|
|
56
|
+
const md = path.join(dir, "SKILL.md");
|
|
57
|
+
return name !== self && !name.startsWith(".") && fs.lstatSync(dir, { throwIfNoEntry: false })?.isDirectory() === true && fs.lstatSync(md, { throwIfNoEntry: false })?.isFile() === true && !isHiddenSkill(fs.readFileSync(md, "utf8"));
|
|
58
|
+
};
|
|
59
|
+
if (install === "all") {
|
|
60
|
+
const all = fs.readdirSync(skillsRoot).filter(invocable).sort();
|
|
61
|
+
if (all.length === 0) throw new Error(`install "all" finds no other model-invocable skill for ${scenarioDir}`);
|
|
62
|
+
return all;
|
|
63
|
+
}
|
|
64
|
+
if (!Array.isArray(install) || install.length === 0) throw new Error(`install must be "all" or a non-empty list of skills in ${scenarioDir}`);
|
|
65
|
+
for (const name of install) if (typeof name !== "string" || name.includes("/") || !invocable(name)) throw new Error(`install names ${JSON.stringify(name)}, not another model-invocable skill under the root, in ${scenarioDir}`);
|
|
66
|
+
return [...new Set(install)].sort();
|
|
67
|
+
}
|
|
41
68
|
function encodeRunNamePart(part) {
|
|
42
69
|
if (!part.includes("--") && !part.startsWith("-") && !part.endsWith("-") && !part.startsWith("~v2~")) return part;
|
|
43
70
|
return `~v2~${encodeURIComponent(part).replaceAll("-", "%2D").replaceAll("~", "%7E")}`;
|
|
@@ -83,6 +110,7 @@ const MARKDOWN_LINK = /\]\(\s*<([^>]+)>|\]\(\s*([^)\s]+)|^ {0,3}\[[^\]]+\]:\s*<?
|
|
|
83
110
|
function linkedSiblings(skillDir) {
|
|
84
111
|
const skillsRoot = path.dirname(skillDir);
|
|
85
112
|
const realRoot = fs.realpathSync(skillsRoot);
|
|
113
|
+
const realSelf = fs.realpathSync(skillDir);
|
|
86
114
|
const self = path.basename(skillDir);
|
|
87
115
|
const found = /* @__PURE__ */ new Set([self]);
|
|
88
116
|
const pending = [skillDir];
|
|
@@ -94,7 +122,8 @@ function linkedSiblings(skillDir) {
|
|
|
94
122
|
const name = rel.split(path.sep)[0];
|
|
95
123
|
if (name === "" || name === ".." || path.isAbsolute(rel) || found.has(name)) continue;
|
|
96
124
|
if (!fs.existsSync(path.join(skillsRoot, name, "SKILL.md"))) continue;
|
|
97
|
-
|
|
125
|
+
const real = fs.realpathSync(path.join(skillsRoot, name));
|
|
126
|
+
if (path.dirname(real) !== realRoot || real === realSelf) continue;
|
|
98
127
|
found.add(name);
|
|
99
128
|
pending.push(path.join(skillsRoot, name));
|
|
100
129
|
}
|
|
@@ -146,9 +175,13 @@ function materialize(s, runDir, harness, control = false) {
|
|
|
146
175
|
}
|
|
147
176
|
const roots = control ? [] : harness === "codex" ? [".claude", ".agents"] : harness === "grok" ? [".grok"] : [".claude"];
|
|
148
177
|
const skillsRoot = path.dirname(s.skillDir);
|
|
149
|
-
const installed = control ? [] : [
|
|
178
|
+
const installed = control ? [] : [.../* @__PURE__ */ new Set([
|
|
179
|
+
s.skill,
|
|
180
|
+
...linkedSiblings(s.skillDir),
|
|
181
|
+
...s.alternatives.flatMap((a) => [a, ...linkedSiblings(path.join(skillsRoot, a))])
|
|
182
|
+
])];
|
|
150
183
|
for (const root of roots) for (const name of installed) {
|
|
151
|
-
const from = path.join(skillsRoot, name);
|
|
184
|
+
const from = fs.realpathSync(path.join(skillsRoot, name));
|
|
152
185
|
fs.cpSync(from, path.join(workdir, root, "skills", name), {
|
|
153
186
|
recursive: true,
|
|
154
187
|
filter: (src) => src !== path.join(from, "evals")
|
|
@@ -177,7 +210,8 @@ function materialize(s, runDir, harness, control = false) {
|
|
|
177
210
|
manifestPath
|
|
178
211
|
};
|
|
179
212
|
}
|
|
180
|
-
function agentProvider(opts, workdir,
|
|
213
|
+
function agentProvider(opts, workdir, skills, paths) {
|
|
214
|
+
const [skill] = skills;
|
|
181
215
|
if (opts.harness === "grok") return {
|
|
182
216
|
id: `file://${paths.grokProviderPath}`,
|
|
183
217
|
config: {
|
|
@@ -211,7 +245,7 @@ function agentProvider(opts, workdir, skill, paths) {
|
|
|
211
245
|
apiKeyRequired: false,
|
|
212
246
|
working_dir: workdir,
|
|
213
247
|
setting_sources: ["project"],
|
|
214
|
-
...opts.control ? {} : { skills
|
|
248
|
+
...opts.control ? {} : { skills },
|
|
215
249
|
permission_mode: "acceptEdits",
|
|
216
250
|
append_allowed_tools: [
|
|
217
251
|
"Read",
|
|
@@ -235,7 +269,7 @@ function buildConfig(s, trials, opts, paths) {
|
|
|
235
269
|
description: `${s.skill}/${s.scenario}`,
|
|
236
270
|
prompts: ["{{task}}"],
|
|
237
271
|
providers: trials.map((t, i) => ({
|
|
238
|
-
...agentProvider(opts, t.workdir, s.skill, paths),
|
|
272
|
+
...agentProvider(opts, t.workdir, [s.skill, ...s.alternatives], paths),
|
|
239
273
|
label: trialLabel(i)
|
|
240
274
|
})),
|
|
241
275
|
defaultTest: { options: {
|
|
@@ -299,7 +333,8 @@ function buildConfig(s, trials, opts, paths) {
|
|
|
299
333
|
metric: "skill-used",
|
|
300
334
|
config: {
|
|
301
335
|
skill: s.skill,
|
|
302
|
-
required: !opts.control && s.criteria.skill_use
|
|
336
|
+
required: !opts.control && (s.criteria.skill_use ?? "required") === "required",
|
|
337
|
+
...!opts.control && s.criteria.skill_use === "forbidden" ? { forbidden: true } : {}
|
|
303
338
|
}
|
|
304
339
|
}]
|
|
305
340
|
}))
|
|
@@ -382,6 +417,7 @@ function handOverToAgent(runDir, trials) {
|
|
|
382
417
|
}
|
|
383
418
|
function generateRun(scenarioDir, opts, paths) {
|
|
384
419
|
const s = loadScenario(scenarioDir);
|
|
420
|
+
if (opts.harness === "grok" && s.criteria.skill_use === "forbidden") throw new Error(`grok cannot evidence a skill load reliably enough for skill_use "forbidden": ${scenarioDir}`);
|
|
385
421
|
const name = runNameFor(scenarioDir, opts.harness, opts.control);
|
|
386
422
|
const runDir = path.join(paths.scratchDir, name);
|
|
387
423
|
fs.rmSync(runDir, {
|
package/dist/skill-evidence.js
CHANGED
|
@@ -34,17 +34,19 @@ function skillEvidence(context) {
|
|
|
34
34
|
const target = real(named);
|
|
35
35
|
return inside && target !== void 0 && installed.has(target);
|
|
36
36
|
})) return `read of the installed ${skill}/SKILL.md`;
|
|
37
|
-
const
|
|
38
|
-
if (calls(metadata?.toolCalls).find((c) => c.name === "Bash" && c.is_error === false && typeof c.
|
|
37
|
+
const opening = [...installed].map((p) => fs.readFileSync(p, "utf8").slice(0, 300).trim()).filter((t) => t.length >= 40);
|
|
38
|
+
if (calls(metadata?.toolCalls).find((c) => c.name === "Bash" && c.is_error === false && typeof c.output === "string" && opening.some((t) => c.output.includes(t)))) return `shell read of the installed ${skill}/SKILL.md`;
|
|
39
39
|
}
|
|
40
40
|
function assertSkillUsed(_output, context) {
|
|
41
41
|
const evidence = skillEvidence(context);
|
|
42
|
-
const
|
|
42
|
+
const forbidden = context.config?.forbidden === true;
|
|
43
|
+
const required = !forbidden && context.config?.required !== false;
|
|
43
44
|
const used = evidence !== void 0;
|
|
45
|
+
const skill = String(context.config?.skill);
|
|
44
46
|
return {
|
|
45
|
-
pass: used || !required,
|
|
47
|
+
pass: forbidden ? !used : used || !required,
|
|
46
48
|
score: used ? 1 : 0,
|
|
47
|
-
reason: used ? `skill used: ${evidence}` : `skill ${
|
|
49
|
+
reason: used ? `skill used: ${evidence}${forbidden ? " (forbidden for this scenario)" : ""}` : `skill ${skill} not loaded${required || forbidden ? "" : " (optional for this scenario)"}`
|
|
48
50
|
};
|
|
49
51
|
}
|
|
50
52
|
//#endregion
|
package/docs/adoption.md
CHANGED
|
@@ -63,7 +63,9 @@ skillcheck sweep # resumes: only scenarios without results
|
|
|
63
63
|
skillcheck summarize # writes .skillcheck/scorecards/<UTC-date>.json
|
|
64
64
|
```
|
|
65
65
|
|
|
66
|
-
|
|
66
|
+
Keep `.skillcheck/scorecards/` as the history, either committed here or in a
|
|
67
|
+
separate evals repository that mounts it as the state directory
|
|
68
|
+
([isolated runs](usage.md#isolated-runs)). Gitignore the rest:
|
|
67
69
|
|
|
68
70
|
```gitignore
|
|
69
71
|
.skillcheck/results/
|
package/docs/scenarios.md
CHANGED
|
@@ -58,7 +58,7 @@ that aggregate, so a run that produces good output without ever loading the
|
|
|
58
58
|
skill still fails. The skill counts as loaded when the harness reports a skill
|
|
59
59
|
call for it, or when the agent completed a read of the installed
|
|
60
60
|
`<config-root>/skills/<skill>/SKILL.md` in its workdir, or a successful shell
|
|
61
|
-
command
|
|
61
|
+
command whose output carries the installed file's opening text.
|
|
62
62
|
Claude Code often reads the file directly instead of calling its Skill tool,
|
|
63
63
|
sometimes from a shell, and either way the same instructions reach its context.
|
|
64
64
|
|
|
@@ -69,6 +69,26 @@ and the result still records whether the skill was loaded. Omitted, it is
|
|
|
69
69
|
reported score is the assert-set's weighted score; skill-used is reported
|
|
70
70
|
separately as a rate across trials.
|
|
71
71
|
|
|
72
|
+
## Trigger scenarios
|
|
73
|
+
|
|
74
|
+
A trigger scenario measures routing: whether a natural prompt loads this skill
|
|
75
|
+
when adjacent skills compete for it. `"install"` names other skills under the
|
|
76
|
+
same root to install beside it, or `"all"` installs every model-invocable one,
|
|
77
|
+
as a plugin does. Alternatives must be real directories, not symlinks. A hidden
|
|
78
|
+
skill is never an alternative and cannot be the skill under test: production
|
|
79
|
+
loads it only on an explicit invocation, so there is no routing to measure.
|
|
80
|
+
|
|
81
|
+
- Positive: `"skill_use": "required"` (the default) with the alternatives
|
|
82
|
+
installed. It fails when the agent routes elsewhere or loads nothing.
|
|
83
|
+
- Near miss: `"skill_use": "forbidden"` on a prompt that belongs to a
|
|
84
|
+
neighbor or to no skill. It fails when this skill loads, and the skill-used
|
|
85
|
+
rate then counts the misfires. Not on Grok, whose load evidence is too thin
|
|
86
|
+
to tell an unloaded skill from an unseen read.
|
|
87
|
+
|
|
88
|
+
Keep the checklist on the outcome the right lane should produce, so a near
|
|
89
|
+
miss that loads nothing but does the work badly still fails. A control run of
|
|
90
|
+
either installs nothing and requires nothing.
|
|
91
|
+
|
|
72
92
|
Write descriptions a judge can check against the deliverable: an observable
|
|
73
93
|
property, not a feeling. Weight the items that would make a reviewer reject the
|
|
74
94
|
work.
|
package/docs/usage.md
CHANGED
|
@@ -107,15 +107,15 @@ would be graded on all of their deliverables.
|
|
|
107
107
|
|
|
108
108
|
A scenario's result aggregates its trials:
|
|
109
109
|
|
|
110
|
-
| Field | Meaning
|
|
111
|
-
| ------------------------------- |
|
|
112
|
-
| `pass` | pass^k: every trial passed. Exit 0 needs this
|
|
113
|
-
| `passes`, `pass_rate` | Trials that passed, as a count and a fraction
|
|
114
|
-
| `score` | Mean weighted checklist score (the assert-set, without skill-used)
|
|
115
|
-
| `score_min` | Lowest trial score
|
|
116
|
-
| `score_spread` | Highest minus lowest trial score
|
|
117
|
-
| `skill_used`, `skill_used_rate` | Trials
|
|
118
|
-
| `noisy` | Trials both passed and failed, or `score_spread` is at least 0.2
|
|
110
|
+
| Field | Meaning |
|
|
111
|
+
| ------------------------------- | ------------------------------------------------------------------------------------------------- |
|
|
112
|
+
| `pass` | pass^k: every trial passed. Exit 0 needs this |
|
|
113
|
+
| `passes`, `pass_rate` | Trials that passed, as a count and a fraction |
|
|
114
|
+
| `score` | Mean weighted checklist score (the assert-set, without skill-used) |
|
|
115
|
+
| `score_min` | Lowest trial score |
|
|
116
|
+
| `score_spread` | Highest minus lowest trial score |
|
|
117
|
+
| `skill_used`, `skill_used_rate` | Trials that loaded the skill, as a count and fraction; on a near-miss scenario these are misfires |
|
|
118
|
+
| `noisy` | Trials both passed and failed, or `score_spread` is at least 0.2 |
|
|
119
119
|
|
|
120
120
|
A trial that errored was never graded, so one errored trial makes the whole
|
|
121
121
|
scenario an ERROR: pass^k over fewer than k trials is not the requested number.
|
|
@@ -281,7 +281,10 @@ becomes `mixed` and per-entry shas remain. A result with no sidecar reduces as
|
|
|
281
281
|
## State
|
|
282
282
|
|
|
283
283
|
`<root>/.skillcheck/` holds `results/`, disposable and safe to gitignore, and
|
|
284
|
-
`scorecards/`,
|
|
284
|
+
`scorecards/`, the history to keep. In the isolated image it is a separate
|
|
285
|
+
writable mount, so it can live outside the repository under test: a separate
|
|
286
|
+
evals repository can pin the skill repository to a commit, keep its scorecards,
|
|
287
|
+
and leave the skill repository with only scenarios and `skillcheck lint`. Scratch workdirs live under the
|
|
285
288
|
system temp dir, outside the root. Nothing is ever written inside the installed
|
|
286
289
|
package.
|
|
287
290
|
|