@uinaf/skillcheck 1.1.0 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -45,7 +45,7 @@ current directory.
45
45
  <root>/skills/<skill>/SKILL.md linted
46
46
  <root>/skills/<skill>/evals/<scenario>/task.md the problem + input files
47
47
  <root>/skills/<skill>/evals/<scenario>/criteria.json the weighted checklist
48
- <root>/.skillcheck/ results, scratch, scorecards
48
+ <root>/.skillcheck/ results, scorecards
49
49
  ```
50
50
 
51
51
  `cli/*/skills/<skill>/` is scanned too, for repos that keep a skill next to the
package/dist/cli.js CHANGED
@@ -2,7 +2,9 @@
2
2
  import { lintSkills } from "./lint.js";
3
3
  import { encodeRunNamePart, generateRun, requiredEvalPackages, resolvePackageDir, runNameFor } from "./scenario.js";
4
4
  import { execFileSync, spawnSync } from "node:child_process";
5
+ import { createHash } from "node:crypto";
5
6
  import fs from "node:fs";
7
+ import os from "node:os";
6
8
  import path from "node:path";
7
9
  import { fileURLToPath } from "node:url";
8
10
  //#region src/cli.ts
@@ -61,9 +63,10 @@ function resolveRoot(flags) {
61
63
  }
62
64
  function stateDirs(root) {
63
65
  const base = path.join(root, ".skillcheck");
66
+ const id = createHash("sha256").update(path.resolve(root)).digest("hex").slice(0, 12);
64
67
  return {
65
68
  results: path.join(base, "results"),
66
- scratch: path.join(base, "scratch"),
69
+ scratch: path.join(fs.realpathSync(os.tmpdir()), `skillcheck-${id}`),
67
70
  scorecards: path.join(base, "scorecards")
68
71
  };
69
72
  }
@@ -80,13 +83,13 @@ function runOptions(flags) {
80
83
  const judgeModel = flags.get("--judge") ?? "claude-opus-5";
81
84
  const judgeEffort = flags.get("--judge-effort");
82
85
  if (judgeEffort !== void 0) {
83
- if (![
86
+ const levels = judgeModel.includes(":") ? [
84
87
  "minimal",
85
88
  "low",
86
89
  "medium",
87
90
  "high"
88
- ].includes(judgeEffort)) throw new Error(`--judge-effort must be minimal, low, medium, or high, got ${judgeEffort}`);
89
- if (!judgeModel.includes(":")) throw new Error("--judge-effort needs a provider-qualified --judge (e.g. openai:chat:gpt-5.6-sol); the Anthropic judge does not take a reasoning effort");
91
+ ] : AGENT_EFFORTS;
92
+ if (!levels.includes(judgeEffort)) throw new Error(`--judge-effort for ${judgeModel} must be ${levels.join(", ")}, got ${judgeEffort}`);
90
93
  }
91
94
  return {
92
95
  harness,
@@ -180,12 +183,12 @@ function classifyRow(raw, stats) {
180
183
  if (typeof res.score !== "number" || typeof res.success !== "boolean") return { error: "promptfoo result carried no usable score" };
181
184
  const components = Array.isArray(res.gradingResult?.componentResults) ? res.gradingResult.componentResults : [];
182
185
  const checklist = components.find((c) => c?.metadata?.assertionSet?.type === "assert-set");
183
- const skillUsed = components.find((c) => c?.assertion?.type === "skill-used");
184
- if (typeof checklist?.score !== "number" || typeof skillUsed?.pass !== "boolean") return { error: "promptfoo result carried no checklist or skill-used verdict" };
186
+ const skillUsed = components.find((c) => c?.assertion?.type === "skill-used" || c?.assertion?.metric === "skill-used");
187
+ if (typeof checklist?.score !== "number" || typeof skillUsed?.score !== "number") return { error: "promptfoo result carried no checklist or skill-used verdict" };
185
188
  return {
186
189
  score: checklist.score,
187
190
  pass: res.success,
188
- skillUsed: skillUsed.pass
191
+ skillUsed: skillUsed.score >= 1
189
192
  };
190
193
  }
191
194
  const NOISY_SPREAD = .2;
@@ -235,12 +238,24 @@ function metaPath(resultPath) {
235
238
  function attemptPath(resultPath) {
236
239
  return `${resultPath}.attempt`;
237
240
  }
241
+ function ensurePrivateDir(dir) {
242
+ fs.mkdirSync(dir, {
243
+ recursive: true,
244
+ mode: 448
245
+ });
246
+ const st = fs.lstatSync(dir);
247
+ const uid = process.getuid?.();
248
+ if (!st.isDirectory() || uid !== void 0 && st.uid !== uid) throw new Error(`scratch dir ${dir} is not a directory owned by this user`);
249
+ if ((st.mode & 63) !== 0) fs.chmodSync(dir, 448);
250
+ }
238
251
  function runScenario(scenarioDir, opts, root) {
239
252
  const dirs = stateDirs(root);
253
+ ensurePrivateDir(dirs.scratch);
240
254
  const { name, configPath, skill, scenario } = generateRun(path.resolve(scenarioDir), opts, {
241
255
  scratchDir: dirs.scratch,
242
256
  transformPath: path.join(here, `transform${selfExt}`),
243
- grokProviderPath: path.join(here, `grok-provider${selfExt}`)
257
+ grokProviderPath: path.join(here, `grok-provider${selfExt}`),
258
+ skillEvidencePath: path.join(here, `skill-evidence${selfExt}`)
244
259
  });
245
260
  fs.mkdirSync(dirs.results, { recursive: true });
246
261
  const resultPath = path.join(dirs.results, `${name}.json`);
@@ -304,7 +319,8 @@ function judgeName(judge) {
304
319
  if (typeof judge === "string") return judge.replace(/^anthropic:messages:/, "");
305
320
  const j = judge;
306
321
  const name = j?.config?.model ?? j?.id;
307
- return typeof name === "string" ? name : "unknown";
322
+ if (typeof name !== "string") return "unknown";
323
+ return j?.config?.effort === void 0 ? name : name.replace(/^anthropic:messages:/, "");
308
324
  }
309
325
  function resultRunConfig(raw, meta, harness) {
310
326
  const m = meta;
@@ -318,7 +334,8 @@ function resultRunConfig(raw, meta, harness) {
318
334
  const r = raw;
319
335
  const agent = r?.config?.providers?.[0]?.config;
320
336
  const judge = r?.config?.defaultTest?.options?.provider;
321
- const judgeEffort = judge?.config?.reasoning_effort;
337
+ const judgeConfig = judge?.config;
338
+ const judgeEffort = judgeConfig?.reasoning_effort ?? judgeConfig?.effort;
322
339
  return {
323
340
  agent_model: typeof agent?.model === "string" ? agent.model : `${harness}-default`,
324
341
  agent_effort: typeof agent?.effort === "string" ? agent.effort : null,
@@ -633,4 +650,4 @@ if (isMainModule()) {
633
650
  }
634
651
  }
635
652
  //#endregion
636
- export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, formatStats, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
653
+ export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, ensurePrivateDir, formatStats, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
package/dist/scenario.js CHANGED
@@ -1,8 +1,8 @@
1
1
  import { createRequire } from "node:module";
2
- import fs from "node:fs";
3
- import path from "node:path";
4
2
  import { createHash } from "node:crypto";
3
+ import fs from "node:fs";
5
4
  import os from "node:os";
5
+ import path from "node:path";
6
6
  //#region src/scenario.ts
7
7
  const DEFAULT_CLAUDE_AGENT = "claude-opus-5";
8
8
  function loadScenario(scenarioDir) {
@@ -12,6 +12,7 @@ function loadScenario(scenarioDir) {
12
12
  const taskMd = fs.readFileSync(path.join(scenarioDir, "task.md"), "utf8");
13
13
  const criteria = JSON.parse(fs.readFileSync(path.join(scenarioDir, "criteria.json"), "utf8"));
14
14
  if (criteria.type !== "weighted_checklist" || !Array.isArray(criteria.checklist) || criteria.checklist.length === 0) throw new Error(`unsupported or empty criteria in ${scenarioDir}`);
15
+ if (criteria.skill_use !== void 0 && criteria.skill_use !== "required" && criteria.skill_use !== "optional") throw new Error(`skill_use must be "required" or "optional" in ${scenarioDir}`);
15
16
  for (const item of criteria.checklist) if (!(typeof item?.name === "string" && item.name.trim() !== "" && typeof item?.description === "string" && item.description.trim() !== "" && Number.isFinite(item?.max_score) && item.max_score > 0)) throw new Error(`invalid checklist item in ${scenarioDir}: ${JSON.stringify(item)}`);
16
17
  const files = [];
17
18
  let prompt = taskMd.replace(/^=+ FILE: (.+?) =+\n([\s\S]*?)\n=+ END FILE =+$/gm, (_, name, content) => {
@@ -179,10 +180,14 @@ function buildConfig(s, trials, opts, paths) {
179
180
  provider: opts.judgeModel.includes(":") ? opts.judgeEffort === void 0 ? opts.judgeModel : {
180
181
  id: opts.judgeModel,
181
182
  config: { reasoning_effort: opts.judgeEffort }
182
- } : process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
183
+ } : process.env.ANTHROPIC_API_KEY ? opts.judgeEffort === void 0 ? `anthropic:messages:${opts.judgeModel}` : {
184
+ id: `anthropic:messages:${opts.judgeModel}`,
185
+ config: { effort: opts.judgeEffort }
186
+ } : {
183
187
  id: "anthropic:claude-agent-sdk",
184
188
  config: {
185
189
  model: opts.judgeModel,
190
+ ...opts.judgeEffort ? { effort: opts.judgeEffort } : {},
186
191
  apiKeyRequired: false,
187
192
  max_turns: 3,
188
193
  output_format: {
@@ -227,8 +232,13 @@ function buildConfig(s, trials, opts, paths) {
227
232
  weight: item.max_score
228
233
  }))
229
234
  }, {
230
- type: "skill-used",
231
- value: s.skill
235
+ type: "javascript",
236
+ value: `file://${paths.skillEvidencePath}`,
237
+ metric: "skill-used",
238
+ config: {
239
+ skill: s.skill,
240
+ required: s.criteria.skill_use !== "optional"
241
+ }
232
242
  }]
233
243
  }))
234
244
  };
@@ -0,0 +1,49 @@
1
+ import fs from "node:fs";
2
+ import path from "node:path";
3
+ //#region src/skill-evidence.ts
4
+ const SKILL_ROOTS = [
5
+ ".claude",
6
+ ".agents",
7
+ ".grok"
8
+ ];
9
+ function real(file) {
10
+ try {
11
+ return fs.realpathSync(file);
12
+ } catch {
13
+ return;
14
+ }
15
+ }
16
+ function skillEvidence(context) {
17
+ const skill = context.config?.skill;
18
+ const workdir = context.vars.workdir;
19
+ if (typeof skill !== "string" || typeof workdir !== "string") return void 0;
20
+ const calls = (value) => Array.isArray(value) ? value.filter((c) => c !== null && typeof c === "object") : [];
21
+ const metadata = context.metadata ?? context.providerResponse?.metadata;
22
+ if (calls(metadata?.skillCalls).find((c) => c.name === skill && c.is_error !== true)) return `skill call ${skill}`;
23
+ const installed = new Set(SKILL_ROOTS.map((root) => real(path.join(workdir, root, "skills", skill, "SKILL.md"))).filter((p) => p !== void 0));
24
+ const root = real(workdir) ?? workdir;
25
+ if (calls(metadata?.toolCalls).find((c) => {
26
+ if (c.name !== "Read" || c.is_error !== false || typeof c.output !== "string") return false;
27
+ const file = c.input?.file_path;
28
+ if (typeof file !== "string") return false;
29
+ const named = path.resolve(workdir, file);
30
+ const inside = [workdir, root].some((w) => {
31
+ const rel = path.relative(w, named);
32
+ return rel !== ".." && !rel.startsWith(`..${path.sep}`) && !path.isAbsolute(rel);
33
+ });
34
+ const target = real(named);
35
+ return inside && target !== void 0 && installed.has(target);
36
+ })) return `read of the installed ${skill}/SKILL.md`;
37
+ }
38
+ function assertSkillUsed(_output, context) {
39
+ const evidence = skillEvidence(context);
40
+ const required = context.config?.required !== false;
41
+ const used = evidence !== void 0;
42
+ return {
43
+ pass: used || !required,
44
+ score: used ? 1 : 0,
45
+ reason: used ? `skill used: ${evidence}` : `skill ${String(context.config?.skill)} not loaded${required ? "" : " (optional for this scenario)"}`
46
+ };
47
+ }
48
+ //#endregion
49
+ export { assertSkillUsed as default, skillEvidence };
package/dist/transform.js CHANGED
@@ -1,9 +1,9 @@
1
+ import { createHash } from "node:crypto";
1
2
  import fs from "node:fs";
2
3
  import path from "node:path";
3
- import { createHash } from "node:crypto";
4
4
  //#region src/transform.ts
5
- const PER_FILE_CAP = 4e3;
6
- const TOTAL_CAP = 24e3;
5
+ const PER_FILE_CAP = 16e3;
6
+ const TOTAL_CAP = 64e3;
7
7
  function safeSlice(text, end) {
8
8
  let cut = text.slice(0, end);
9
9
  const last = cut.charCodeAt(cut.length - 1);
package/docs/adoption.md CHANGED
@@ -63,7 +63,6 @@ Commit `.skillcheck/scorecards/`. Gitignore the rest:
63
63
 
64
64
  ```gitignore
65
65
  .skillcheck/results/
66
- .skillcheck/scratch/
67
66
  ```
68
67
 
69
68
  A scorecard is only comparable against the tree it graded and the configuration
package/docs/scenarios.md CHANGED
@@ -54,7 +54,16 @@ item needs a non-empty `name` and `description` and a positive `max_score`.
54
54
  Each item becomes one `llm-rubric` assertion weighted by `max_score`, inside an
55
55
  assert-set with threshold 0.7. A separate `skill-used` assertion sits outside
56
56
  that aggregate, so a run that produces good output without ever loading the
57
- skill still fails. There is no test-level threshold: both must pass. The
57
+ skill still fails. The skill counts as loaded when the harness reports a skill
58
+ call for it, or when the agent completed a read of the installed
59
+ `<config-root>/skills/<skill>/SKILL.md` in its workdir. Claude Code often reads
60
+ the file directly instead of calling its Skill tool, and either way the same
61
+ instructions reach its context.
62
+
63
+ An out-of-lane scenario, where the right answer is to decline the skill, sets
64
+ `"skill_use": "optional"` in `criteria.json`. The assertion then always passes,
65
+ and the result still records whether the skill was loaded. Omitted, it is
66
+ `"required"`. There is no test-level threshold: both must pass. The
58
67
  reported score is the assert-set's weighted score; skill-used is reported
59
68
  separately as a rate across trials.
60
69
 
@@ -62,14 +71,27 @@ Write descriptions a judge can check against the deliverable: an observable
62
71
  property, not a feeling. Weight the items that would make a reviewer reject the
63
72
  work.
64
73
 
74
+ On the Claude harness the agent can read, search, and write files in its
75
+ workdir, but has no shell, so it cannot install, build, test, or reach the
76
+ network. The shell stays off because the agent runs on the operator's machine
77
+ with the operator's credentials. Codex runs shell commands inside its
78
+ `workspace-write` sandbox, and Grok follows its own CLI permission mode. Write criteria a shell-less agent can meet, so one scenario
79
+ scores comparably across harnesses: a checklist item that requires live proof,
80
+ such as a frozen lockfile or a verified release, fails on Claude. Grade whether
81
+ the deliverable names the checks it could not run and hands them over
82
+ precisely.
83
+
84
+ Name a specific tool or version only when the skill teaches it. Otherwise grade
85
+ the property the tool provides, so an equivalent approach passes.
86
+
65
87
  ## What the judge sees
66
88
 
67
89
  The agent's final message, plus every file in the workdir that differs from the
68
90
  pre-run manifest. Unchanged inputs are omitted; deleted inputs, unreadable
69
91
  files, and non-regular files are named rather than read.
70
92
 
71
- Sections are sorted by path, each file is capped at 4,000 characters and the
72
- appended total at 24,000, with truncation stated inline. Very large outputs make
93
+ Sections are sorted by path, each file is capped at 16,000 characters and the
94
+ appended total at 64,000, with truncation stated inline. Very large outputs make
73
95
  rubric judges return nothing at all, which is why the caps exist. Keep fixtures
74
96
  small enough that the deliverable fits.
75
97
 
@@ -84,8 +106,10 @@ frontmatter block; body text mentioning the key does not count.
84
106
 
85
107
  ## The workdir
86
108
 
87
- Per run, under `<root>/.skillcheck/scratch/<name>/`, rebuilt from scratch each
88
- time. The skill under test is installed where the harness discovers skills:
109
+ Per run, under a scratch directory in the system temp dir
110
+ (`skillcheck-<hash of the root>/<name>/trial-<n>/`), rebuilt from scratch each
111
+ time. It stays outside the root so the agent cannot reach the skill's source,
112
+ its evals, or the repository's own agent guidance. The skill under test is installed where the harness discovers skills:
89
113
  `.claude/skills/<skill>/`, plus `.agents/skills/<skill>/` on codex, or
90
114
  `.grok/skills/<skill>/` on Grok, with its `evals/` directory excluded, so
91
115
  criteria never leak into the agent's context.
package/docs/usage.md CHANGED
@@ -38,7 +38,7 @@ skillcheck run <scenario-dir> --harness grok
38
38
  skillcheck run <scenario-dir> --trials 3 --agent-effort medium
39
39
  ```
40
40
 
41
- Materializes the scenario into `<root>/.skillcheck/scratch/<name>/trial-<n>/workdir`,
41
+ Materializes the scenario into a temp-dir workdir per trial ([workdir](scenarios.md#the-workdir)),
42
42
  installs the skill under test into that workdir, drives the agent, and grades
43
43
  the files it wrote. Exit 0 means pass, 1 means graded fail, and 2 means error.
44
44
  Exit 2 covers missing usable promptfoo output or optional eval peers. The
@@ -109,9 +109,16 @@ skillcheck run <scenario-dir> --judge openai:chat:gpt-5.6-sol --judge-effort hig
109
109
 
110
110
  A provider-qualified judge authenticates through that provider's own env
111
111
  (`OPENAI_API_KEY`, plus `OPENAI_BASE_URL` for a gateway) and is recorded
112
- verbatim in the scorecard's `judge_model` column. `--judge-effort`
113
- (minimal|low|medium|high) sets `reasoning_effort` and requires a
114
- provider-qualified judge; the Anthropic judge does not take one.
112
+ verbatim in the scorecard's `judge_model` column. For it, `--judge-effort`
113
+ (minimal|low|medium|high) sets `reasoning_effort`. For a bare Claude judge,
114
+ `--judge-effort` takes Claude's levels (low|medium|high|xhigh|max) and is passed
115
+ as `effort` on either Anthropic path; the SDK judge starts Claude Code with
116
+ `--effort`:
117
+
118
+ ```sh
119
+ skillcheck run <scenario-dir> --agent claude-opus-5-5 --agent-effort medium \
120
+ --judge claude-opus-5-5 --judge-effort high --trials 3
121
+ ```
115
122
 
116
123
  ## Sweep
117
124
 
@@ -211,9 +218,10 @@ becomes `mixed` and per-entry shas remain. A result with no sidecar reduces as
211
218
 
212
219
  ## State
213
220
 
214
- `<root>/.skillcheck/` holds `scratch/` and `results/`, both disposable and safe
215
- to gitignore, and `scorecards/`, which is meant to be committed. Nothing is ever
216
- written inside the installed package.
221
+ `<root>/.skillcheck/` holds `results/`, disposable and safe to gitignore, and
222
+ `scorecards/`, which is meant to be committed. Scratch workdirs live under the
223
+ system temp dir, outside the root. Nothing is ever written inside the installed
224
+ package.
217
225
 
218
226
  ## Auth
219
227
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "1.1.0",
3
+ "version": "1.3.0",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {