@uinaf/skillcheck 1.2.0 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -45,7 +45,7 @@ current directory.
45
45
  <root>/skills/<skill>/SKILL.md linted
46
46
  <root>/skills/<skill>/evals/<scenario>/task.md the problem + input files
47
47
  <root>/skills/<skill>/evals/<scenario>/criteria.json the weighted checklist
48
- <root>/.skillcheck/ results, scratch, scorecards
48
+ <root>/.skillcheck/ results, scorecards
49
49
  ```
50
50
 
51
51
  `cli/*/skills/<skill>/` is scanned too, for repos that keep a skill next to the
package/dist/cli.js CHANGED
@@ -2,7 +2,9 @@
2
2
  import { lintSkills } from "./lint.js";
3
3
  import { encodeRunNamePart, generateRun, requiredEvalPackages, resolvePackageDir, runNameFor } from "./scenario.js";
4
4
  import { execFileSync, spawnSync } from "node:child_process";
5
+ import { createHash } from "node:crypto";
5
6
  import fs from "node:fs";
7
+ import os from "node:os";
6
8
  import path from "node:path";
7
9
  import { fileURLToPath } from "node:url";
8
10
  //#region src/cli.ts
@@ -61,9 +63,10 @@ function resolveRoot(flags) {
61
63
  }
62
64
  function stateDirs(root) {
63
65
  const base = path.join(root, ".skillcheck");
66
+ const id = createHash("sha256").update(path.resolve(root)).digest("hex").slice(0, 12);
64
67
  return {
65
68
  results: path.join(base, "results"),
66
- scratch: path.join(base, "scratch"),
69
+ scratch: path.join(fs.realpathSync(os.tmpdir()), `skillcheck-${id}`),
67
70
  scorecards: path.join(base, "scorecards")
68
71
  };
69
72
  }
@@ -180,12 +183,12 @@ function classifyRow(raw, stats) {
180
183
  if (typeof res.score !== "number" || typeof res.success !== "boolean") return { error: "promptfoo result carried no usable score" };
181
184
  const components = Array.isArray(res.gradingResult?.componentResults) ? res.gradingResult.componentResults : [];
182
185
  const checklist = components.find((c) => c?.metadata?.assertionSet?.type === "assert-set");
183
- const skillUsed = components.find((c) => c?.assertion?.type === "skill-used");
184
- if (typeof checklist?.score !== "number" || typeof skillUsed?.pass !== "boolean") return { error: "promptfoo result carried no checklist or skill-used verdict" };
186
+ const skillUsed = components.find((c) => c?.assertion?.type === "skill-used" || c?.assertion?.metric === "skill-used");
187
+ if (typeof checklist?.score !== "number" || typeof skillUsed?.score !== "number") return { error: "promptfoo result carried no checklist or skill-used verdict" };
185
188
  return {
186
189
  score: checklist.score,
187
190
  pass: res.success,
188
- skillUsed: skillUsed.pass
191
+ skillUsed: skillUsed.score >= 1
189
192
  };
190
193
  }
191
194
  const NOISY_SPREAD = .2;
@@ -235,12 +238,24 @@ function metaPath(resultPath) {
235
238
  function attemptPath(resultPath) {
236
239
  return `${resultPath}.attempt`;
237
240
  }
241
+ function ensurePrivateDir(dir) {
242
+ fs.mkdirSync(dir, {
243
+ recursive: true,
244
+ mode: 448
245
+ });
246
+ const st = fs.lstatSync(dir);
247
+ const uid = process.getuid?.();
248
+ if (!st.isDirectory() || uid !== void 0 && st.uid !== uid) throw new Error(`scratch dir ${dir} is not a directory owned by this user`);
249
+ if ((st.mode & 63) !== 0) fs.chmodSync(dir, 448);
250
+ }
238
251
  function runScenario(scenarioDir, opts, root) {
239
252
  const dirs = stateDirs(root);
253
+ ensurePrivateDir(dirs.scratch);
240
254
  const { name, configPath, skill, scenario } = generateRun(path.resolve(scenarioDir), opts, {
241
255
  scratchDir: dirs.scratch,
242
256
  transformPath: path.join(here, `transform${selfExt}`),
243
- grokProviderPath: path.join(here, `grok-provider${selfExt}`)
257
+ grokProviderPath: path.join(here, `grok-provider${selfExt}`),
258
+ skillEvidencePath: path.join(here, `skill-evidence${selfExt}`)
244
259
  });
245
260
  fs.mkdirSync(dirs.results, { recursive: true });
246
261
  const resultPath = path.join(dirs.results, `${name}.json`);
@@ -635,4 +650,4 @@ if (isMainModule()) {
635
650
  }
636
651
  }
637
652
  //#endregion
638
- export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, formatStats, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
653
+ export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, ensurePrivateDir, formatStats, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
package/dist/scenario.js CHANGED
@@ -1,8 +1,8 @@
1
1
  import { createRequire } from "node:module";
2
- import fs from "node:fs";
3
- import path from "node:path";
4
2
  import { createHash } from "node:crypto";
3
+ import fs from "node:fs";
5
4
  import os from "node:os";
5
+ import path from "node:path";
6
6
  //#region src/scenario.ts
7
7
  const DEFAULT_CLAUDE_AGENT = "claude-opus-5";
8
8
  function loadScenario(scenarioDir) {
@@ -12,6 +12,7 @@ function loadScenario(scenarioDir) {
12
12
  const taskMd = fs.readFileSync(path.join(scenarioDir, "task.md"), "utf8");
13
13
  const criteria = JSON.parse(fs.readFileSync(path.join(scenarioDir, "criteria.json"), "utf8"));
14
14
  if (criteria.type !== "weighted_checklist" || !Array.isArray(criteria.checklist) || criteria.checklist.length === 0) throw new Error(`unsupported or empty criteria in ${scenarioDir}`);
15
+ if (criteria.skill_use !== void 0 && criteria.skill_use !== "required" && criteria.skill_use !== "optional") throw new Error(`skill_use must be "required" or "optional" in ${scenarioDir}`);
15
16
  for (const item of criteria.checklist) if (!(typeof item?.name === "string" && item.name.trim() !== "" && typeof item?.description === "string" && item.description.trim() !== "" && Number.isFinite(item?.max_score) && item.max_score > 0)) throw new Error(`invalid checklist item in ${scenarioDir}: ${JSON.stringify(item)}`);
16
17
  const files = [];
17
18
  let prompt = taskMd.replace(/^=+ FILE: (.+?) =+\n([\s\S]*?)\n=+ END FILE =+$/gm, (_, name, content) => {
@@ -231,8 +232,13 @@ function buildConfig(s, trials, opts, paths) {
231
232
  weight: item.max_score
232
233
  }))
233
234
  }, {
234
- type: "skill-used",
235
- value: s.skill
235
+ type: "javascript",
236
+ value: `file://${paths.skillEvidencePath}`,
237
+ metric: "skill-used",
238
+ config: {
239
+ skill: s.skill,
240
+ required: s.criteria.skill_use !== "optional"
241
+ }
236
242
  }]
237
243
  }))
238
244
  };
@@ -0,0 +1,49 @@
1
+ import fs from "node:fs";
2
+ import path from "node:path";
3
+ //#region src/skill-evidence.ts
4
+ const SKILL_ROOTS = [
5
+ ".claude",
6
+ ".agents",
7
+ ".grok"
8
+ ];
9
+ function real(file) {
10
+ try {
11
+ return fs.realpathSync(file);
12
+ } catch {
13
+ return;
14
+ }
15
+ }
16
+ function skillEvidence(context) {
17
+ const skill = context.config?.skill;
18
+ const workdir = context.vars.workdir;
19
+ if (typeof skill !== "string" || typeof workdir !== "string") return void 0;
20
+ const calls = (value) => Array.isArray(value) ? value.filter((c) => c !== null && typeof c === "object") : [];
21
+ const metadata = context.metadata ?? context.providerResponse?.metadata;
22
+ if (calls(metadata?.skillCalls).find((c) => c.name === skill && c.is_error !== true)) return `skill call ${skill}`;
23
+ const installed = new Set(SKILL_ROOTS.map((root) => real(path.join(workdir, root, "skills", skill, "SKILL.md"))).filter((p) => p !== void 0));
24
+ const root = real(workdir) ?? workdir;
25
+ if (calls(metadata?.toolCalls).find((c) => {
26
+ if (c.name !== "Read" || c.is_error !== false || typeof c.output !== "string") return false;
27
+ const file = c.input?.file_path;
28
+ if (typeof file !== "string") return false;
29
+ const named = path.resolve(workdir, file);
30
+ const inside = [workdir, root].some((w) => {
31
+ const rel = path.relative(w, named);
32
+ return rel !== ".." && !rel.startsWith(`..${path.sep}`) && !path.isAbsolute(rel);
33
+ });
34
+ const target = real(named);
35
+ return inside && target !== void 0 && installed.has(target);
36
+ })) return `read of the installed ${skill}/SKILL.md`;
37
+ }
38
+ function assertSkillUsed(_output, context) {
39
+ const evidence = skillEvidence(context);
40
+ const required = context.config?.required !== false;
41
+ const used = evidence !== void 0;
42
+ return {
43
+ pass: used || !required,
44
+ score: used ? 1 : 0,
45
+ reason: used ? `skill used: ${evidence}` : `skill ${String(context.config?.skill)} not loaded${required ? "" : " (optional for this scenario)"}`
46
+ };
47
+ }
48
+ //#endregion
49
+ export { assertSkillUsed as default, skillEvidence };
package/dist/transform.js CHANGED
@@ -1,9 +1,9 @@
1
+ import { createHash } from "node:crypto";
1
2
  import fs from "node:fs";
2
3
  import path from "node:path";
3
- import { createHash } from "node:crypto";
4
4
  //#region src/transform.ts
5
- const PER_FILE_CAP = 4e3;
6
- const TOTAL_CAP = 24e3;
5
+ const PER_FILE_CAP = 16e3;
6
+ const TOTAL_CAP = 64e3;
7
7
  function safeSlice(text, end) {
8
8
  let cut = text.slice(0, end);
9
9
  const last = cut.charCodeAt(cut.length - 1);
package/docs/adoption.md CHANGED
@@ -63,7 +63,6 @@ Commit `.skillcheck/scorecards/`. Gitignore the rest:
63
63
 
64
64
  ```gitignore
65
65
  .skillcheck/results/
66
- .skillcheck/scratch/
67
66
  ```
68
67
 
69
68
  A scorecard is only comparable against the tree it graded and the configuration
package/docs/scenarios.md CHANGED
@@ -54,7 +54,16 @@ item needs a non-empty `name` and `description` and a positive `max_score`.
54
54
  Each item becomes one `llm-rubric` assertion weighted by `max_score`, inside an
55
55
  assert-set with threshold 0.7. A separate `skill-used` assertion sits outside
56
56
  that aggregate, so a run that produces good output without ever loading the
57
- skill still fails. There is no test-level threshold: both must pass. The
57
+ skill still fails. The skill counts as loaded when the harness reports a skill
58
+ call for it, or when the agent completed a read of the installed
59
+ `<config-root>/skills/<skill>/SKILL.md` in its workdir. Claude Code often reads
60
+ the file directly instead of calling its Skill tool, and either way the same
61
+ instructions reach its context.
62
+
63
+ An out-of-lane scenario, where the right answer is to decline the skill, sets
64
+ `"skill_use": "optional"` in `criteria.json`. The assertion then always passes,
65
+ and the result still records whether the skill was loaded. Omitted, it is
66
+ `"required"`. There is no test-level threshold: both must pass. The
58
67
  reported score is the assert-set's weighted score; skill-used is reported
59
68
  separately as a rate across trials.
60
69
 
@@ -62,14 +71,27 @@ Write descriptions a judge can check against the deliverable: an observable
62
71
  property, not a feeling. Weight the items that would make a reviewer reject the
63
72
  work.
64
73
 
74
+ On the Claude harness the agent can read, search, and write files in its
75
+ workdir, but has no shell, so it cannot install, build, test, or reach the
76
+ network. The shell stays off because the agent runs on the operator's machine
77
+ with the operator's credentials. Codex runs shell commands inside its
78
+ `workspace-write` sandbox, and Grok follows its own CLI permission mode. Write criteria a shell-less agent can meet, so one scenario
79
+ scores comparably across harnesses: a checklist item that requires live proof,
80
+ such as a frozen lockfile or a verified release, fails on Claude. Grade whether
81
+ the deliverable names the checks it could not run and hands them over
82
+ precisely.
83
+
84
+ Name a specific tool or version only when the skill teaches it. Otherwise grade
85
+ the property the tool provides, so an equivalent approach passes.
86
+
65
87
  ## What the judge sees
66
88
 
67
89
  The agent's final message, plus every file in the workdir that differs from the
68
90
  pre-run manifest. Unchanged inputs are omitted; deleted inputs, unreadable
69
91
  files, and non-regular files are named rather than read.
70
92
 
71
- Sections are sorted by path, each file is capped at 4,000 characters and the
72
- appended total at 24,000, with truncation stated inline. Very large outputs make
93
+ Sections are sorted by path, each file is capped at 16,000 characters and the
94
+ appended total at 64,000, with truncation stated inline. Very large outputs make
73
95
  rubric judges return nothing at all, which is why the caps exist. Keep fixtures
74
96
  small enough that the deliverable fits.
75
97
 
@@ -84,8 +106,10 @@ frontmatter block; body text mentioning the key does not count.
84
106
 
85
107
  ## The workdir
86
108
 
87
- Per run, under `<root>/.skillcheck/scratch/<name>/`, rebuilt from scratch each
88
- time. The skill under test is installed where the harness discovers skills:
109
+ Per run, under a scratch directory in the system temp dir
110
+ (`skillcheck-<hash of the root>/<name>/trial-<n>/`), rebuilt from scratch each
111
+ time. It stays outside the root so the agent cannot reach the skill's source,
112
+ its evals, or the repository's own agent guidance. The skill under test is installed where the harness discovers skills:
89
113
  `.claude/skills/<skill>/`, plus `.agents/skills/<skill>/` on codex, or
90
114
  `.grok/skills/<skill>/` on Grok, with its `evals/` directory excluded, so
91
115
  criteria never leak into the agent's context.
package/docs/usage.md CHANGED
@@ -38,7 +38,7 @@ skillcheck run <scenario-dir> --harness grok
38
38
  skillcheck run <scenario-dir> --trials 3 --agent-effort medium
39
39
  ```
40
40
 
41
- Materializes the scenario into `<root>/.skillcheck/scratch/<name>/trial-<n>/workdir`,
41
+ Materializes the scenario into a temp-dir workdir per trial ([workdir](scenarios.md#the-workdir)),
42
42
  installs the skill under test into that workdir, drives the agent, and grades
43
43
  the files it wrote. Exit 0 means pass, 1 means graded fail, and 2 means error.
44
44
  Exit 2 covers missing usable promptfoo output or optional eval peers. The
@@ -218,9 +218,10 @@ becomes `mixed` and per-entry shas remain. A result with no sidecar reduces as
218
218
 
219
219
  ## State
220
220
 
221
- `<root>/.skillcheck/` holds `scratch/` and `results/`, both disposable and safe
222
- to gitignore, and `scorecards/`, which is meant to be committed. Nothing is ever
223
- written inside the installed package.
221
+ `<root>/.skillcheck/` holds `results/`, disposable and safe to gitignore, and
222
+ `scorecards/`, which is meant to be committed. Scratch workdirs live under the
223
+ system temp dir, outside the root. Nothing is ever written inside the installed
224
+ package.
224
225
 
225
226
  ## Auth
226
227
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "1.2.0",
3
+ "version": "1.3.0",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {