@uinaf/skillcheck 1.6.1 → 1.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/scenario.js CHANGED
@@ -14,7 +14,11 @@ function loadScenario(scenarioDir) {
14
14
  const taskMd = fs.readFileSync(path.join(scenarioDir, "task.md"), "utf8");
15
15
  const criteria = JSON.parse(fs.readFileSync(path.join(scenarioDir, "criteria.json"), "utf8"));
16
16
  if (criteria.type !== "weighted_checklist" || !Array.isArray(criteria.checklist) || criteria.checklist.length === 0) throw new Error(`unsupported or empty criteria in ${scenarioDir}`);
17
- if (criteria.skill_use !== void 0 && criteria.skill_use !== "required" && criteria.skill_use !== "optional") throw new Error(`skill_use must be "required" or "optional" in ${scenarioDir}`);
17
+ if (criteria.skill_use !== void 0 && ![
18
+ "required",
19
+ "optional",
20
+ "forbidden"
21
+ ].includes(criteria.skill_use)) throw new Error(`skill_use must be "required", "optional", or "forbidden" in ${scenarioDir}`);
18
22
  for (const item of criteria.checklist) if (!(typeof item?.name === "string" && item.name.trim() !== "" && typeof item?.description === "string" && item.description.trim() !== "" && Number.isFinite(item?.max_score) && item.max_score > 0)) throw new Error(`invalid checklist item in ${scenarioDir}: ${JSON.stringify(item)}`);
19
23
  const files = [];
20
24
  const task = taskMd.replace(FILE_BLOCK, (_, name, content) => {
@@ -26,18 +30,41 @@ function loadScenario(scenarioDir) {
26
30
  });
27
31
  let prompt = task;
28
32
  const skillDir = path.resolve(scenarioDir, "../..");
29
- if (isHiddenSkill(fs.readFileSync(path.join(skillDir, "SKILL.md"), "utf8"))) prompt = `Use the ${skill} skill for this task.\n\n${prompt}`;
33
+ if (isHiddenSkill(fs.readFileSync(path.join(skillDir, "SKILL.md"), "utf8"))) {
34
+ if (criteria.skill_use === "forbidden" || criteria.install !== void 0) throw new Error(`a hidden skill is always invoked, so it has no routing to test: ${scenarioDir}`);
35
+ prompt = `Use the ${skill} skill for this task.\n\n${prompt}`;
36
+ }
37
+ const alternatives = resolveInstall(skillDir, criteria.install, scenarioDir);
30
38
  return {
31
39
  skill,
32
40
  scenario,
33
41
  name: `${skill}--${scenario}`,
34
42
  skillDir,
43
+ alternatives,
35
44
  prompt,
36
45
  task,
37
46
  files,
38
47
  criteria
39
48
  };
40
49
  }
50
+ function resolveInstall(skillDir, install, scenarioDir) {
51
+ if (install === void 0) return [];
52
+ const skillsRoot = path.dirname(skillDir);
53
+ const self = path.basename(skillDir);
54
+ const invocable = (name) => {
55
+ const dir = path.join(skillsRoot, name);
56
+ const md = path.join(dir, "SKILL.md");
57
+ return name !== self && !name.startsWith(".") && fs.lstatSync(dir, { throwIfNoEntry: false })?.isDirectory() === true && fs.lstatSync(md, { throwIfNoEntry: false })?.isFile() === true && !isHiddenSkill(fs.readFileSync(md, "utf8"));
58
+ };
59
+ if (install === "all") {
60
+ const all = fs.readdirSync(skillsRoot).filter(invocable).sort();
61
+ if (all.length === 0) throw new Error(`install "all" finds no other model-invocable skill for ${scenarioDir}`);
62
+ return all;
63
+ }
64
+ if (!Array.isArray(install) || install.length === 0) throw new Error(`install must be "all" or a non-empty list of skills in ${scenarioDir}`);
65
+ for (const name of install) if (typeof name !== "string" || name.includes("/") || !invocable(name)) throw new Error(`install names ${JSON.stringify(name)}, not another model-invocable skill under the root, in ${scenarioDir}`);
66
+ return [...new Set(install)].sort();
67
+ }
41
68
  function encodeRunNamePart(part) {
42
69
  if (!part.includes("--") && !part.startsWith("-") && !part.endsWith("-") && !part.startsWith("~v2~")) return part;
43
70
  return `~v2~${encodeURIComponent(part).replaceAll("-", "%2D").replaceAll("~", "%7E")}`;
@@ -83,6 +110,7 @@ const MARKDOWN_LINK = /\]\(\s*<([^>]+)>|\]\(\s*([^)\s]+)|^ {0,3}\[[^\]]+\]:\s*<?
83
110
  function linkedSiblings(skillDir) {
84
111
  const skillsRoot = path.dirname(skillDir);
85
112
  const realRoot = fs.realpathSync(skillsRoot);
113
+ const realSelf = fs.realpathSync(skillDir);
86
114
  const self = path.basename(skillDir);
87
115
  const found = /* @__PURE__ */ new Set([self]);
88
116
  const pending = [skillDir];
@@ -94,7 +122,8 @@ function linkedSiblings(skillDir) {
94
122
  const name = rel.split(path.sep)[0];
95
123
  if (name === "" || name === ".." || path.isAbsolute(rel) || found.has(name)) continue;
96
124
  if (!fs.existsSync(path.join(skillsRoot, name, "SKILL.md"))) continue;
97
- if (path.dirname(fs.realpathSync(path.join(skillsRoot, name))) !== realRoot) continue;
125
+ const real = fs.realpathSync(path.join(skillsRoot, name));
126
+ if (path.dirname(real) !== realRoot || real === realSelf) continue;
98
127
  found.add(name);
99
128
  pending.push(path.join(skillsRoot, name));
100
129
  }
@@ -146,9 +175,13 @@ function materialize(s, runDir, harness, control = false) {
146
175
  }
147
176
  const roots = control ? [] : harness === "codex" ? [".claude", ".agents"] : harness === "grok" ? [".grok"] : [".claude"];
148
177
  const skillsRoot = path.dirname(s.skillDir);
149
- const installed = control ? [] : [s.skill, ...linkedSiblings(s.skillDir)];
178
+ const installed = control ? [] : [.../* @__PURE__ */ new Set([
179
+ s.skill,
180
+ ...linkedSiblings(s.skillDir),
181
+ ...s.alternatives.flatMap((a) => [a, ...linkedSiblings(path.join(skillsRoot, a))])
182
+ ])];
150
183
  for (const root of roots) for (const name of installed) {
151
- const from = path.join(skillsRoot, name);
184
+ const from = fs.realpathSync(path.join(skillsRoot, name));
152
185
  fs.cpSync(from, path.join(workdir, root, "skills", name), {
153
186
  recursive: true,
154
187
  filter: (src) => src !== path.join(from, "evals")
@@ -177,7 +210,8 @@ function materialize(s, runDir, harness, control = false) {
177
210
  manifestPath
178
211
  };
179
212
  }
180
- function agentProvider(opts, workdir, skill, paths) {
213
+ function agentProvider(opts, workdir, skills, paths) {
214
+ const [skill] = skills;
181
215
  if (opts.harness === "grok") return {
182
216
  id: `file://${paths.grokProviderPath}`,
183
217
  config: {
@@ -211,7 +245,7 @@ function agentProvider(opts, workdir, skill, paths) {
211
245
  apiKeyRequired: false,
212
246
  working_dir: workdir,
213
247
  setting_sources: ["project"],
214
- ...opts.control ? {} : { skills: [skill] },
248
+ ...opts.control ? {} : { skills },
215
249
  permission_mode: "acceptEdits",
216
250
  append_allowed_tools: [
217
251
  "Read",
@@ -235,7 +269,7 @@ function buildConfig(s, trials, opts, paths) {
235
269
  description: `${s.skill}/${s.scenario}`,
236
270
  prompts: ["{{task}}"],
237
271
  providers: trials.map((t, i) => ({
238
- ...agentProvider(opts, t.workdir, s.skill, paths),
272
+ ...agentProvider(opts, t.workdir, [s.skill, ...s.alternatives], paths),
239
273
  label: trialLabel(i)
240
274
  })),
241
275
  defaultTest: { options: {
@@ -299,7 +333,8 @@ function buildConfig(s, trials, opts, paths) {
299
333
  metric: "skill-used",
300
334
  config: {
301
335
  skill: s.skill,
302
- required: !opts.control && s.criteria.skill_use !== "optional"
336
+ required: !opts.control && (s.criteria.skill_use ?? "required") === "required",
337
+ ...!opts.control && s.criteria.skill_use === "forbidden" ? { forbidden: true } : {}
303
338
  }
304
339
  }]
305
340
  }))
@@ -382,6 +417,7 @@ function handOverToAgent(runDir, trials) {
382
417
  }
383
418
  function generateRun(scenarioDir, opts, paths) {
384
419
  const s = loadScenario(scenarioDir);
420
+ if (opts.harness === "grok" && s.criteria.skill_use === "forbidden") throw new Error(`grok cannot evidence a skill load reliably enough for skill_use "forbidden": ${scenarioDir}`);
385
421
  const name = runNameFor(scenarioDir, opts.harness, opts.control);
386
422
  const runDir = path.join(paths.scratchDir, name);
387
423
  fs.rmSync(runDir, {
@@ -34,17 +34,19 @@ function skillEvidence(context) {
34
34
  const target = real(named);
35
35
  return inside && target !== void 0 && installed.has(target);
36
36
  })) return `read of the installed ${skill}/SKILL.md`;
37
- const nameLine = new RegExp(`^name:\\s*["']?${skill.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")}["']?\\s*$`, "m");
38
- if (calls(metadata?.toolCalls).find((c) => c.name === "Bash" && c.is_error === false && typeof c.input?.command === "string" && c.input.command.includes("SKILL.md") && typeof c.output === "string" && nameLine.test(c.output))) return `shell read of the installed ${skill}/SKILL.md`;
37
+ const opening = [...installed].map((p) => fs.readFileSync(p, "utf8").slice(0, 300).trim()).filter((t) => t.length >= 40);
38
+ if (calls(metadata?.toolCalls).find((c) => c.name === "Bash" && c.is_error === false && typeof c.output === "string" && opening.some((t) => c.output.includes(t)))) return `shell read of the installed ${skill}/SKILL.md`;
39
39
  }
40
40
  function assertSkillUsed(_output, context) {
41
41
  const evidence = skillEvidence(context);
42
- const required = context.config?.required !== false;
42
+ const forbidden = context.config?.forbidden === true;
43
+ const required = !forbidden && context.config?.required !== false;
43
44
  const used = evidence !== void 0;
45
+ const skill = String(context.config?.skill);
44
46
  return {
45
- pass: used || !required,
47
+ pass: forbidden ? !used : used || !required,
46
48
  score: used ? 1 : 0,
47
- reason: used ? `skill used: ${evidence}` : `skill ${String(context.config?.skill)} not loaded${required ? "" : " (optional for this scenario)"}`
49
+ reason: used ? `skill used: ${evidence}${forbidden ? " (forbidden for this scenario)" : ""}` : `skill ${skill} not loaded${required || forbidden ? "" : " (optional for this scenario)"}`
48
50
  };
49
51
  }
50
52
  //#endregion
package/docs/adoption.md CHANGED
@@ -63,7 +63,9 @@ skillcheck sweep # resumes: only scenarios without results
63
63
  skillcheck summarize # writes .skillcheck/scorecards/<UTC-date>.json
64
64
  ```
65
65
 
66
- Commit `.skillcheck/scorecards/`. Gitignore the rest:
66
+ Keep `.skillcheck/scorecards/` as the history, either committed here or in a
67
+ separate evals repository that mounts it as the state directory
68
+ ([isolated runs](usage.md#isolated-runs)). Gitignore the rest:
67
69
 
68
70
  ```gitignore
69
71
  .skillcheck/results/
package/docs/scenarios.md CHANGED
@@ -58,7 +58,7 @@ that aggregate, so a run that produces good output without ever loading the
58
58
  skill still fails. The skill counts as loaded when the harness reports a skill
59
59
  call for it, or when the agent completed a read of the installed
60
60
  `<config-root>/skills/<skill>/SKILL.md` in its workdir, or a successful shell
61
- command that names `SKILL.md` and printed this skill's frontmatter `name`.
61
+ command whose output carries the installed file's opening text.
62
62
  Claude Code often reads the file directly instead of calling its Skill tool,
63
63
  sometimes from a shell, and either way the same instructions reach its context.
64
64
 
@@ -69,6 +69,26 @@ and the result still records whether the skill was loaded. Omitted, it is
69
69
  reported score is the assert-set's weighted score; skill-used is reported
70
70
  separately as a rate across trials.
71
71
 
72
+ ## Trigger scenarios
73
+
74
+ A trigger scenario measures routing: whether a natural prompt loads this skill
75
+ when adjacent skills compete for it. `"install"` names other skills under the
76
+ same root to install beside it, or `"all"` installs every model-invocable one,
77
+ as a plugin does. Alternatives must be real directories, not symlinks. A hidden
78
+ skill is never an alternative and cannot be the skill under test: production
79
+ loads it only on an explicit invocation, so there is no routing to measure.
80
+
81
+ - Positive: `"skill_use": "required"` (the default) with the alternatives
82
+ installed. It fails when the agent routes elsewhere or loads nothing.
83
+ - Near miss: `"skill_use": "forbidden"` on a prompt that belongs to a
84
+ neighbor or to no skill. It fails when this skill loads, and the skill-used
85
+ rate then counts the misfires. Not on Grok, whose load evidence is too thin
86
+ to tell an unloaded skill from an unseen read.
87
+
88
+ Keep the checklist on the outcome the right lane should produce, so a near
89
+ miss that loads nothing but does the work badly still fails. A control run of
90
+ either installs nothing and requires nothing.
91
+
72
92
  Write descriptions a judge can check against the deliverable: an observable
73
93
  property, not a feeling. Weight the items that would make a reviewer reject the
74
94
  work.
package/docs/usage.md CHANGED
@@ -107,15 +107,15 @@ would be graded on all of their deliverables.
107
107
 
108
108
  A scenario's result aggregates its trials:
109
109
 
110
- | Field | Meaning |
111
- | ------------------------------- | ------------------------------------------------------------------- |
112
- | `pass` | pass^k: every trial passed. Exit 0 needs this |
113
- | `passes`, `pass_rate` | Trials that passed, as a count and a fraction |
114
- | `score` | Mean weighted checklist score (the assert-set, without skill-used) |
115
- | `score_min` | Lowest trial score |
116
- | `score_spread` | Highest minus lowest trial score |
117
- | `skill_used`, `skill_used_rate` | Trials whose `skill-used` assertion passed, as a count and fraction |
118
- | `noisy` | Trials both passed and failed, or `score_spread` is at least 0.2 |
110
+ | Field | Meaning |
111
+ | ------------------------------- | ------------------------------------------------------------------------------------------------- |
112
+ | `pass` | pass^k: every trial passed. Exit 0 needs this |
113
+ | `passes`, `pass_rate` | Trials that passed, as a count and a fraction |
114
+ | `score` | Mean weighted checklist score (the assert-set, without skill-used) |
115
+ | `score_min` | Lowest trial score |
116
+ | `score_spread` | Highest minus lowest trial score |
117
+ | `skill_used`, `skill_used_rate` | Trials that loaded the skill, as a count and fraction; on a near-miss scenario these are misfires |
118
+ | `noisy` | Trials both passed and failed, or `score_spread` is at least 0.2 |
119
119
 
120
120
  A trial that errored was never graded, so one errored trial makes the whole
121
121
  scenario an ERROR: pass^k over fewer than k trials is not the requested number.
@@ -281,7 +281,10 @@ becomes `mixed` and per-entry shas remain. A result with no sidecar reduces as
281
281
  ## State
282
282
 
283
283
  `<root>/.skillcheck/` holds `results/`, disposable and safe to gitignore, and
284
- `scorecards/`, which is meant to be committed. Scratch workdirs live under the
284
+ `scorecards/`, the history to keep. In the isolated image it is a separate
285
+ writable mount, so it can live outside the repository under test: a separate
286
+ evals repository can pin the skill repository to a commit, keep its scorecards,
287
+ and leave the skill repository with only scenarios and `skillcheck lint`. Scratch workdirs live under the
285
288
  system temp dir, outside the root. Nothing is ever written inside the installed
286
289
  package.
287
290
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "1.6.1",
3
+ "version": "1.7.0",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {