@uinaf/skillcheck 1.2.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -45,7 +45,7 @@ current directory.
45
45
  <root>/skills/<skill>/SKILL.md linted
46
46
  <root>/skills/<skill>/evals/<scenario>/task.md the problem + input files
47
47
  <root>/skills/<skill>/evals/<scenario>/criteria.json the weighted checklist
48
- <root>/.skillcheck/ results, scratch, scorecards
48
+ <root>/.skillcheck/ results, scorecards
49
49
  ```
50
50
 
51
51
  `cli/*/skills/<skill>/` is scanned too, for repos that keep a skill next to the
package/dist/cli.js CHANGED
@@ -2,7 +2,9 @@
2
2
  import { lintSkills } from "./lint.js";
3
3
  import { encodeRunNamePart, generateRun, requiredEvalPackages, resolvePackageDir, runNameFor } from "./scenario.js";
4
4
  import { execFileSync, spawnSync } from "node:child_process";
5
+ import { createHash } from "node:crypto";
5
6
  import fs from "node:fs";
7
+ import os from "node:os";
6
8
  import path from "node:path";
7
9
  import { fileURLToPath } from "node:url";
8
10
  //#region src/cli.ts
@@ -48,7 +50,7 @@ function parseArgs(argv) {
48
50
  const v = argv[++i];
49
51
  if (v === void 0 || v.startsWith("--")) throw new Error(`${a} needs a value`);
50
52
  flags.set(a, v);
51
- } else if (a === "--all" || a === "--allow-mixed") flags.set(a, true);
53
+ } else if (a === "--all" || a === "--allow-mixed" || a === "--control") flags.set(a, true);
52
54
  else throw new Error(`unknown flag: ${a}`);
53
55
  }
54
56
  return {
@@ -61,9 +63,10 @@ function resolveRoot(flags) {
61
63
  }
62
64
  function stateDirs(root) {
63
65
  const base = path.join(root, ".skillcheck");
66
+ const id = createHash("sha256").update(path.resolve(root)).digest("hex").slice(0, 12);
64
67
  return {
65
68
  results: path.join(base, "results"),
66
- scratch: path.join(base, "scratch"),
69
+ scratch: path.join(fs.realpathSync(os.tmpdir()), `skillcheck-${id}`),
67
70
  scorecards: path.join(base, "scorecards")
68
71
  };
69
72
  }
@@ -95,6 +98,7 @@ function runOptions(flags) {
95
98
  judgeModel,
96
99
  judgeEffort,
97
100
  maxTurns: flags.has("--max-turns") ? parsePositiveInt("--max-turns", flags.get("--max-turns")) : void 0,
101
+ control: flags.get("--control") === true,
98
102
  trials: flags.has("--trials") ? parsePositiveInt("--trials", flags.get("--trials")) : 1
99
103
  };
100
104
  }
@@ -104,7 +108,8 @@ function runConfigOf(opts) {
104
108
  agent_effort: opts.agentEffort ?? null,
105
109
  judge_model: opts.judgeModel,
106
110
  judge_effort: opts.judgeEffort ?? null,
107
- trials: opts.trials ?? 1
111
+ trials: opts.trials ?? 1,
112
+ agent_access: "online"
108
113
  };
109
114
  }
110
115
  function configKey(c) {
@@ -113,12 +118,13 @@ function configKey(c) {
113
118
  c.agent_effort ?? null,
114
119
  c.judge_model,
115
120
  c.judge_effort ?? null,
116
- c.trials ?? 1
121
+ c.trials ?? 1,
122
+ c.agent_access ?? "offline"
117
123
  ]);
118
124
  }
119
125
  function describeConfig(c) {
120
126
  const effort = (e) => e ? `@${e}` : "";
121
- return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}`;
127
+ return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}, ${c.agent_access ?? "offline"}`;
122
128
  }
123
129
  function assertUniformConfig(entries, allowMixed) {
124
130
  if (allowMixed) return;
@@ -180,12 +186,12 @@ function classifyRow(raw, stats) {
180
186
  if (typeof res.score !== "number" || typeof res.success !== "boolean") return { error: "promptfoo result carried no usable score" };
181
187
  const components = Array.isArray(res.gradingResult?.componentResults) ? res.gradingResult.componentResults : [];
182
188
  const checklist = components.find((c) => c?.metadata?.assertionSet?.type === "assert-set");
183
- const skillUsed = components.find((c) => c?.assertion?.type === "skill-used");
184
- if (typeof checklist?.score !== "number" || typeof skillUsed?.pass !== "boolean") return { error: "promptfoo result carried no checklist or skill-used verdict" };
189
+ const skillUsed = components.find((c) => c?.assertion?.type === "skill-used" || c?.assertion?.metric === "skill-used");
190
+ if (typeof checklist?.score !== "number" || typeof skillUsed?.score !== "number") return { error: "promptfoo result carried no checklist or skill-used verdict" };
185
191
  return {
186
192
  score: checklist.score,
187
193
  pass: res.success,
188
- skillUsed: skillUsed.pass
194
+ skillUsed: skillUsed.score >= 1
189
195
  };
190
196
  }
191
197
  const NOISY_SPREAD = .2;
@@ -235,19 +241,32 @@ function metaPath(resultPath) {
235
241
  function attemptPath(resultPath) {
236
242
  return `${resultPath}.attempt`;
237
243
  }
244
+ function ensurePrivateDir(dir) {
245
+ fs.mkdirSync(dir, {
246
+ recursive: true,
247
+ mode: 448
248
+ });
249
+ const st = fs.lstatSync(dir);
250
+ const uid = process.getuid?.();
251
+ if (!st.isDirectory() || uid !== void 0 && st.uid !== uid) throw new Error(`scratch dir ${dir} is not a directory owned by this user`);
252
+ if ((st.mode & 63) !== 0) fs.chmodSync(dir, 448);
253
+ }
238
254
  function runScenario(scenarioDir, opts, root) {
239
255
  const dirs = stateDirs(root);
256
+ ensurePrivateDir(dirs.scratch);
240
257
  const { name, configPath, skill, scenario } = generateRun(path.resolve(scenarioDir), opts, {
241
258
  scratchDir: dirs.scratch,
242
259
  transformPath: path.join(here, `transform${selfExt}`),
243
- grokProviderPath: path.join(here, `grok-provider${selfExt}`)
260
+ grokProviderPath: path.join(here, `grok-provider${selfExt}`),
261
+ skillEvidencePath: path.join(here, `skill-evidence${selfExt}`)
244
262
  });
245
263
  fs.mkdirSync(dirs.results, { recursive: true });
246
264
  const resultPath = path.join(dirs.results, `${name}.json`);
247
265
  const identity = {
248
266
  skill,
249
267
  scenario,
250
- harness: opts.harness
268
+ harness: opts.harness,
269
+ variant: opts.control ? "control" : "skill"
251
270
  };
252
271
  fs.writeFileSync(attemptPath(resultPath), JSON.stringify(identity) + "\n");
253
272
  fs.rmSync(resultPath, { force: true });
@@ -314,7 +333,8 @@ function resultRunConfig(raw, meta, harness) {
314
333
  agent_effort: m.agent_effort ?? null,
315
334
  judge_model: m.judge_model,
316
335
  judge_effort: m.judge_effort ?? null,
317
- trials: m.trials ?? 1
336
+ trials: m.trials ?? 1,
337
+ agent_access: m.agent_access === "online" ? "online" : "offline"
318
338
  };
319
339
  const r = raw;
320
340
  const agent = r?.config?.providers?.[0]?.config;
@@ -326,7 +346,8 @@ function resultRunConfig(raw, meta, harness) {
326
346
  agent_effort: typeof agent?.effort === "string" ? agent.effort : null,
327
347
  judge_model: judgeName(judge),
328
348
  judge_effort: typeof judgeEffort === "string" ? judgeEffort : null,
329
- trials: r?.results?.results?.length ?? 1
349
+ trials: r?.results?.results?.length ?? 1,
350
+ agent_access: "offline"
330
351
  };
331
352
  }
332
353
  function readJson(file) {
@@ -356,7 +377,7 @@ function discoverScenarios(root) {
356
377
  }
357
378
  return found.sort();
358
379
  }
359
- const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N]";
380
+ const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N] [--control]";
360
381
  function cmdRun(argv) {
361
382
  const { positional, flags } = parseArgs(argv);
362
383
  if (positional.length !== 1) fail(`usage: skillcheck run <scenario-dir> ${RUN_FLAGS}`);
@@ -381,7 +402,7 @@ function cmdSweep(argv) {
381
402
  const wanted = configKey(runConfigOf(opts));
382
403
  let passed = 0, failed = 0, errored = 0, skipped = 0;
383
404
  for (const dir of discoverScenarios(root)) {
384
- const name = runNameFor(dir, opts.harness);
405
+ const name = runNameFor(dir, opts.harness, opts.control);
385
406
  const resultPath = path.join(resultsDir, `${name}.json`);
386
407
  if (!all && fs.existsSync(resultPath) && !fs.existsSync(attemptPath(resultPath))) {
387
408
  const raw = readJson(resultPath);
@@ -416,7 +437,8 @@ function resultIdentity(file, dir) {
416
437
  if (identity !== null && typeof identity === "object" && "skill" in identity && typeof identity.skill === "string" && "scenario" in identity && typeof identity.scenario === "string" && "harness" in identity && (identity.harness === "claude" || identity.harness === "codex" || identity.harness === "grok")) return {
417
438
  skill: identity.skill,
418
439
  scenario: identity.scenario,
419
- harness: identity.harness
440
+ harness: identity.harness,
441
+ variant: "variant" in identity && identity.variant === "control" ? "control" : "skill"
420
442
  };
421
443
  } catch {}
422
444
  if (/^~v3~[0-9a-f]{64}$/.test(base)) throw new Error(`missing identity metadata for ${file}`);
@@ -429,13 +451,17 @@ function resultIdentity(file, dir) {
429
451
  return part;
430
452
  }
431
453
  };
432
- const suffix = base.match(/--(codex|grok|cursor)$/);
454
+ const unsuffixed = base.replace(/--control$/, "");
455
+ const variant = unsuffixed !== base && unsuffixed.includes("--") ? "control" : "skill";
456
+ const stem = variant === "control" ? unsuffixed : base;
457
+ const suffix = stem.match(/--(codex|grok|cursor)$/);
433
458
  const harness = suffix === null ? "claude" : suffix[1];
434
- const [skill, ...rest] = base.replace(/--(codex|grok|cursor)$/, "").split("--");
459
+ const [skill, ...rest] = stem.replace(/--(codex|grok|cursor)$/, "").split("--");
435
460
  return {
436
461
  skill: decode(skill),
437
462
  scenario: decode(rest.join("--")),
438
- harness
463
+ harness,
464
+ variant
439
465
  };
440
466
  }
441
467
  function reduceResults(dir, allowMixed) {
@@ -460,7 +486,7 @@ function reduceResults(dir, allowMixed) {
460
486
  const raw = readJson(path.join(dir, f));
461
487
  const base = f.replace(/\.json$/, "");
462
488
  const meta = readJson(path.join(dir, `${base}.meta.json`));
463
- const { skill, scenario, harness } = resultIdentity(f, dir);
489
+ const { skill, scenario, harness, variant } = resultIdentity(f, dir);
464
490
  const config = resultRunConfig(raw, meta, harness);
465
491
  const verdict = classifyResult(raw, config.trials);
466
492
  if ("error" in verdict) {
@@ -472,7 +498,8 @@ function reduceResults(dir, allowMixed) {
472
498
  const key = entryKey({
473
499
  skill,
474
500
  scenario,
475
- harness
501
+ harness,
502
+ variant
476
503
  });
477
504
  gradedAt.set(key, Math.max(gradedAt.get(key) ?? 0, fs.statSync(path.join(dir, f)).mtimeMs));
478
505
  const sha = typeof meta?.skills_tree_sha === "string" ? meta.skills_tree_sha : "unattested";
@@ -482,6 +509,7 @@ function reduceResults(dir, allowMixed) {
482
509
  skill,
483
510
  scenario,
484
511
  harness,
512
+ variant,
485
513
  skills_tree_sha: sha,
486
514
  ...stats,
487
515
  ...config,
@@ -501,7 +529,8 @@ function entryKey(e) {
501
529
  return [
502
530
  e.skill,
503
531
  e.scenario,
504
- e.harness
532
+ e.harness,
533
+ e.variant ?? "skill"
505
534
  ].join("\0");
506
535
  }
507
536
  function mergeScorecard(existing, fresh) {
@@ -558,27 +587,47 @@ function cmdSummarize(argv) {
558
587
  scenarios: merged.entries
559
588
  };
560
589
  fs.writeFileSync(out, JSON.stringify(scorecard, null, 2) + "\n");
561
- console.log(`${out}: ${merged.entries.length} scenario(s), ${merged.entries.filter((e) => e.pass).length} passing, ${skipped.length} skipped file(s)`);
590
+ const scenarios = merged.entries.filter((e) => e.variant !== "control");
591
+ console.log(`${out}: ${scenarios.length} scenario(s), ${scenarios.filter((e) => e.pass).length} passing, ${merged.entries.length - scenarios.length} control(s), ${skipped.length} skipped file(s)`);
562
592
  if (existing.length > 0) console.log(`merged into today's scorecard: ${entries.length} from this run, ${merged.carried} carried over`);
563
593
  console.log(`\n${formatSkillTable(summarizeSkills(merged.entries))}`);
564
- for (const e of merged.entries.filter((x) => x.noisy)) console.log(`NOISY ${e.skill}/${e.scenario} (${e.harness}) ${formatStats(e)}`);
594
+ for (const e of merged.entries.filter((x) => x.noisy)) {
595
+ const tag = e.variant === "control" ? ", control" : "";
596
+ console.log(`NOISY ${e.skill}/${e.scenario} (${e.harness}${tag}) ${formatStats(e)}`);
597
+ }
598
+ for (const s of summarizeSkills(merged.entries)) for (const scenario of s.no_lift) console.log(`NO LIFT ${s.skill}/${scenario} (${s.harness}): passes without the skill`);
565
599
  }
566
600
  function summarizeSkills(entries) {
567
601
  const groups = /* @__PURE__ */ new Map();
602
+ const controls = /* @__PURE__ */ new Map();
568
603
  for (const e of entries) {
569
604
  const key = `${e.skill}\0${e.harness}`;
570
- groups.set(key, [...groups.get(key) ?? [], e]);
605
+ if (e.variant === "control") controls.set(`${key}\0${e.scenario}`, e);
606
+ else groups.set(key, [...groups.get(key) ?? [], e]);
571
607
  }
572
608
  const mean = (xs) => round4(xs.reduce((a, b) => a + b, 0) / xs.length);
573
- return [...groups.values()].map((rows) => ({
574
- skill: rows[0].skill,
575
- harness: rows[0].harness,
576
- scenarios: rows.length,
577
- pass_all: rows.filter((r) => r.pass).length,
578
- pass_rate: mean(rows.map((r) => r.pass_rate ?? (r.pass ? 1 : 0))),
579
- score: mean(rows.map((r) => r.score)),
580
- noisy: rows.filter((r) => r.noisy).map((r) => r.scenario)
581
- }));
609
+ return [...groups.entries()].map(([key, rows]) => {
610
+ const paired = rows.flatMap((r) => {
611
+ const c = controls.get(`${key}\0${r.scenario}`);
612
+ return c === void 0 ? [] : [{
613
+ skill: r,
614
+ control: c
615
+ }];
616
+ });
617
+ const controlScore = paired.length === 0 ? null : mean(paired.map((p) => p.control.score));
618
+ return {
619
+ skill: rows[0].skill,
620
+ harness: rows[0].harness,
621
+ scenarios: rows.length,
622
+ pass_all: rows.filter((r) => r.pass).length,
623
+ pass_rate: mean(rows.map((r) => r.pass_rate ?? (r.pass ? 1 : 0))),
624
+ score: mean(rows.map((r) => r.score)),
625
+ noisy: rows.filter((r) => r.noisy).map((r) => r.scenario),
626
+ control_score: controlScore,
627
+ lift: controlScore === null ? null : round4(mean(paired.map((p) => p.skill.score)) - controlScore),
628
+ no_lift: paired.filter((p) => p.control.pass).map((p) => p.skill.scenario)
629
+ };
630
+ });
582
631
  }
583
632
  function formatSkillTable(rows) {
584
633
  const table = [[
@@ -588,7 +637,9 @@ function formatSkillTable(rows) {
588
637
  "pass^k",
589
638
  "pass rate",
590
639
  "score",
591
- "noisy"
640
+ "noisy",
641
+ "control",
642
+ "lift"
592
643
  ], ...rows.map((r) => [
593
644
  r.skill,
594
645
  r.harness,
@@ -596,7 +647,9 @@ function formatSkillTable(rows) {
596
647
  `${r.pass_all}/${r.scenarios}`,
597
648
  r.pass_rate.toFixed(2),
598
649
  r.score.toFixed(2),
599
- String(r.noisy.length)
650
+ String(r.noisy.length),
651
+ r.control_score === null ? "-" : r.control_score.toFixed(2),
652
+ r.lift === null ? "-" : `${r.lift >= 0 ? "+" : ""}${r.lift.toFixed(2)}`
600
653
  ])];
601
654
  const widths = table[0].map((_, i) => Math.max(...table.map((row) => row[i].length)));
602
655
  return table.map((row) => row.map((c, i) => c.padEnd(widths[i])).join(" ").trimEnd()).join("\n");
@@ -635,4 +688,4 @@ if (isMainModule()) {
635
688
  }
636
689
  }
637
690
  //#endregion
638
- export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, formatStats, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
691
+ export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, ensurePrivateDir, formatStats, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
@@ -27,7 +27,6 @@ var GrokProvider = class {
27
27
  "streaming-json",
28
28
  "--permission-mode",
29
29
  "acceptEdits",
30
- "--disable-web-search",
31
30
  "--no-subagents",
32
31
  ...model ? ["--model", model] : [],
33
32
  "-p",
package/dist/scenario.js CHANGED
@@ -1,8 +1,8 @@
1
1
  import { createRequire } from "node:module";
2
- import fs from "node:fs";
3
- import path from "node:path";
4
2
  import { createHash } from "node:crypto";
3
+ import fs from "node:fs";
5
4
  import os from "node:os";
5
+ import path from "node:path";
6
6
  //#region src/scenario.ts
7
7
  const DEFAULT_CLAUDE_AGENT = "claude-opus-5";
8
8
  function loadScenario(scenarioDir) {
@@ -12,15 +12,17 @@ function loadScenario(scenarioDir) {
12
12
  const taskMd = fs.readFileSync(path.join(scenarioDir, "task.md"), "utf8");
13
13
  const criteria = JSON.parse(fs.readFileSync(path.join(scenarioDir, "criteria.json"), "utf8"));
14
14
  if (criteria.type !== "weighted_checklist" || !Array.isArray(criteria.checklist) || criteria.checklist.length === 0) throw new Error(`unsupported or empty criteria in ${scenarioDir}`);
15
+ if (criteria.skill_use !== void 0 && criteria.skill_use !== "required" && criteria.skill_use !== "optional") throw new Error(`skill_use must be "required" or "optional" in ${scenarioDir}`);
15
16
  for (const item of criteria.checklist) if (!(typeof item?.name === "string" && item.name.trim() !== "" && typeof item?.description === "string" && item.description.trim() !== "" && Number.isFinite(item?.max_score) && item.max_score > 0)) throw new Error(`invalid checklist item in ${scenarioDir}: ${JSON.stringify(item)}`);
16
17
  const files = [];
17
- let prompt = taskMd.replace(/^=+ FILE: (.+?) =+\n([\s\S]*?)\n=+ END FILE =+$/gm, (_, name, content) => {
18
+ const task = taskMd.replace(/^=+ FILE: (.+?) =+\n([\s\S]*?)\n=+ END FILE =+$/gm, (_, name, content) => {
18
19
  files.push({
19
20
  name: name.trim(),
20
21
  content: content + "\n"
21
22
  });
22
23
  return `(Input file \`${name.trim()}\` is available in your working directory.)`;
23
24
  });
25
+ let prompt = task;
24
26
  const skillDir = path.resolve(scenarioDir, "../..");
25
27
  if (isHiddenSkill(fs.readFileSync(path.join(skillDir, "SKILL.md"), "utf8"))) prompt = `Use the ${skill} skill for this task.\n\n${prompt}`;
26
28
  return {
@@ -29,6 +31,7 @@ function loadScenario(scenarioDir) {
29
31
  name: `${skill}--${scenario}`,
30
32
  skillDir,
31
33
  prompt,
34
+ task,
32
35
  files,
33
36
  criteria
34
37
  };
@@ -37,13 +40,18 @@ function encodeRunNamePart(part) {
37
40
  if (!part.includes("--") && !part.startsWith("-") && !part.endsWith("-") && !part.startsWith("~v2~")) return part;
38
41
  return `~v2~${encodeURIComponent(part).replaceAll("-", "%2D").replaceAll("~", "%7E")}`;
39
42
  }
40
- function runNameFor(scenarioDir, harness) {
43
+ function runNameFor(scenarioDir, harness, control = false) {
41
44
  const m = path.resolve(scenarioDir).match(/skills\/([^/]+)\/evals\/([^/]+)$/);
42
45
  if (!m) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
43
46
  const name = `${encodeRunNamePart(m[1])}--${encodeRunNamePart(m[2])}`;
44
- const full = harness === "claude" ? name : `${name}--${harness}`;
47
+ const full = `${harness === "claude" ? name : `${name}--${harness}`}${control ? "--control" : ""}`;
45
48
  if (Buffer.byteLength(`${full}.json.attempt`) <= 255) return full;
46
- return `~v3~${createHash("sha256").update(JSON.stringify([
49
+ return `~v3~${createHash("sha256").update(JSON.stringify(control ? [
50
+ m[1],
51
+ m[2],
52
+ harness,
53
+ "control"
54
+ ] : [
47
55
  m[1],
48
56
  m[2],
49
57
  harness
@@ -71,7 +79,7 @@ const RESERVED = /* @__PURE__ */ new Set([
71
79
  ".grok",
72
80
  "node_modules"
73
81
  ]);
74
- function materialize(s, runDir, harness) {
82
+ function materialize(s, runDir, harness, control = false) {
75
83
  const workdir = path.join(runDir, "workdir");
76
84
  fs.rmSync(runDir, {
77
85
  recursive: true,
@@ -95,7 +103,7 @@ function materialize(s, runDir, harness) {
95
103
  fs.mkdirSync(path.dirname(dest), { recursive: true });
96
104
  fs.writeFileSync(dest, content);
97
105
  }
98
- const roots = harness === "codex" ? [".claude", ".agents"] : harness === "grok" ? [".grok"] : [".claude"];
106
+ const roots = control ? [] : harness === "codex" ? [".claude", ".agents"] : harness === "grok" ? [".grok"] : [".claude"];
99
107
  for (const root of roots) fs.cpSync(s.skillDir, path.join(workdir, root, "skills", s.skill), {
100
108
  recursive: true,
101
109
  filter: (src) => path.basename(src) !== "evals"
@@ -140,7 +148,9 @@ function agentProvider(opts, workdir, skill, paths) {
140
148
  skip_git_repo_check: true,
141
149
  enable_streaming: true,
142
150
  sandbox_mode: "workspace-write",
143
- cli_env: { CODEX_HOME: process.env.CODEX_HOME ?? path.join(os.homedir(), ".codex") }
151
+ network_access_enabled: true,
152
+ web_search_enabled: true,
153
+ cli_env: { CODEX_HOME: path.join(workdir, "..", "..", "codex-home") }
144
154
  }
145
155
  };
146
156
  return {
@@ -151,14 +161,17 @@ function agentProvider(opts, workdir, skill, paths) {
151
161
  apiKeyRequired: false,
152
162
  working_dir: workdir,
153
163
  setting_sources: ["project"],
154
- skills: [skill],
164
+ ...opts.control ? {} : { skills: [skill] },
155
165
  permission_mode: "acceptEdits",
156
166
  append_allowed_tools: [
157
167
  "Read",
158
168
  "Write",
159
169
  "Edit",
160
170
  "Glob",
161
- "Grep"
171
+ "Grep",
172
+ "Bash",
173
+ "WebFetch",
174
+ "WebSearch"
162
175
  ],
163
176
  max_turns: opts.maxTurns ?? 50
164
177
  }
@@ -218,7 +231,7 @@ function buildConfig(s, trials, opts, paths) {
218
231
  description: s.criteria.context,
219
232
  providers: [trialLabel(i)],
220
233
  vars: {
221
- task: s.prompt,
234
+ task: opts.control ? s.task : s.prompt,
222
235
  workdir: t.workdir,
223
236
  manifest: t.manifestPath
224
237
  },
@@ -231,8 +244,13 @@ function buildConfig(s, trials, opts, paths) {
231
244
  weight: item.max_score
232
245
  }))
233
246
  }, {
234
- type: "skill-used",
235
- value: s.skill
247
+ type: "javascript",
248
+ value: `file://${paths.skillEvidencePath}`,
249
+ metric: "skill-used",
250
+ config: {
251
+ skill: s.skill,
252
+ required: !opts.control && s.criteria.skill_use !== "optional"
253
+ }
236
254
  }]
237
255
  }))
238
256
  };
@@ -259,15 +277,31 @@ function requiredEvalPackages(opts, hasAnthropicKey) {
259
277
  if (opts.harness === "codex") pkgs.push("@openai/codex-sdk");
260
278
  return pkgs;
261
279
  }
280
+ function privateCodexHome(dir) {
281
+ const source = process.env.CODEX_HOME ?? path.join(os.homedir(), ".codex");
282
+ fs.rmSync(dir, {
283
+ recursive: true,
284
+ force: true
285
+ });
286
+ fs.mkdirSync(dir, {
287
+ recursive: true,
288
+ mode: 448
289
+ });
290
+ for (const file of ["config.toml", "auth.json"]) {
291
+ const from = path.join(source, file);
292
+ if (fs.existsSync(from)) fs.symlinkSync(from, path.join(dir, file));
293
+ }
294
+ }
262
295
  function generateRun(scenarioDir, opts, paths) {
263
296
  const s = loadScenario(scenarioDir);
264
- const name = runNameFor(scenarioDir, opts.harness);
297
+ const name = runNameFor(scenarioDir, opts.harness, opts.control);
265
298
  const runDir = path.join(paths.scratchDir, name);
266
299
  fs.rmSync(runDir, {
267
300
  recursive: true,
268
301
  force: true
269
302
  });
270
- const trials = Array.from({ length: opts.trials ?? 1 }, (_, i) => materialize(s, path.join(runDir, trialLabel(i)), opts.harness));
303
+ const trials = Array.from({ length: opts.trials ?? 1 }, (_, i) => materialize(s, path.join(runDir, trialLabel(i)), opts.harness, opts.control));
304
+ if (opts.harness === "codex") privateCodexHome(path.join(runDir, "codex-home"));
271
305
  const sdkDir = sdkNodeModulesDir();
272
306
  if (sdkDir !== void 0) {
273
307
  const link = path.join(runDir, "node_modules");
@@ -288,4 +322,4 @@ function generateRun(scenarioDir, opts, paths) {
288
322
  };
289
323
  }
290
324
  //#endregion
291
- export { DEFAULT_CLAUDE_AGENT, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
325
+ export { DEFAULT_CLAUDE_AGENT, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, privateCodexHome, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
@@ -0,0 +1,49 @@
1
+ import fs from "node:fs";
2
+ import path from "node:path";
3
+ //#region src/skill-evidence.ts
4
+ const SKILL_ROOTS = [
5
+ ".claude",
6
+ ".agents",
7
+ ".grok"
8
+ ];
9
+ function real(file) {
10
+ try {
11
+ return fs.realpathSync(file);
12
+ } catch {
13
+ return;
14
+ }
15
+ }
16
+ function skillEvidence(context) {
17
+ const skill = context.config?.skill;
18
+ const workdir = context.vars.workdir;
19
+ if (typeof skill !== "string" || typeof workdir !== "string") return void 0;
20
+ const calls = (value) => Array.isArray(value) ? value.filter((c) => c !== null && typeof c === "object") : [];
21
+ const metadata = context.metadata ?? context.providerResponse?.metadata;
22
+ if (calls(metadata?.skillCalls).find((c) => c.name === skill && c.is_error !== true)) return `skill call ${skill}`;
23
+ const installed = new Set(SKILL_ROOTS.map((root) => real(path.join(workdir, root, "skills", skill, "SKILL.md"))).filter((p) => p !== void 0));
24
+ const root = real(workdir) ?? workdir;
25
+ if (calls(metadata?.toolCalls).find((c) => {
26
+ if (c.name !== "Read" || c.is_error !== false || typeof c.output !== "string") return false;
27
+ const file = c.input?.file_path;
28
+ if (typeof file !== "string") return false;
29
+ const named = path.resolve(workdir, file);
30
+ const inside = [workdir, root].some((w) => {
31
+ const rel = path.relative(w, named);
32
+ return rel !== ".." && !rel.startsWith(`..${path.sep}`) && !path.isAbsolute(rel);
33
+ });
34
+ const target = real(named);
35
+ return inside && target !== void 0 && installed.has(target);
36
+ })) return `read of the installed ${skill}/SKILL.md`;
37
+ }
38
+ function assertSkillUsed(_output, context) {
39
+ const evidence = skillEvidence(context);
40
+ const required = context.config?.required !== false;
41
+ const used = evidence !== void 0;
42
+ return {
43
+ pass: used || !required,
44
+ score: used ? 1 : 0,
45
+ reason: used ? `skill used: ${evidence}` : `skill ${String(context.config?.skill)} not loaded${required ? "" : " (optional for this scenario)"}`
46
+ };
47
+ }
48
+ //#endregion
49
+ export { assertSkillUsed as default, skillEvidence };
package/dist/transform.js CHANGED
@@ -1,9 +1,9 @@
1
+ import { createHash } from "node:crypto";
1
2
  import fs from "node:fs";
2
3
  import path from "node:path";
3
- import { createHash } from "node:crypto";
4
4
  //#region src/transform.ts
5
- const PER_FILE_CAP = 4e3;
6
- const TOTAL_CAP = 24e3;
5
+ const PER_FILE_CAP = 16e3;
6
+ const TOTAL_CAP = 64e3;
7
7
  function safeSlice(text, end) {
8
8
  let cut = text.slice(0, end);
9
9
  const last = cut.charCodeAt(cut.length - 1);
package/docs/adoption.md CHANGED
@@ -63,7 +63,6 @@ Commit `.skillcheck/scorecards/`. Gitignore the rest:
63
63
 
64
64
  ```gitignore
65
65
  .skillcheck/results/
66
- .skillcheck/scratch/
67
66
  ```
68
67
 
69
68
  A scorecard is only comparable against the tree it graded and the configuration
package/docs/scenarios.md CHANGED
@@ -54,7 +54,16 @@ item needs a non-empty `name` and `description` and a positive `max_score`.
54
54
  Each item becomes one `llm-rubric` assertion weighted by `max_score`, inside an
55
55
  assert-set with threshold 0.7. A separate `skill-used` assertion sits outside
56
56
  that aggregate, so a run that produces good output without ever loading the
57
- skill still fails. There is no test-level threshold: both must pass. The
57
+ skill still fails. The skill counts as loaded when the harness reports a skill
58
+ call for it, or when the agent completed a read of the installed
59
+ `<config-root>/skills/<skill>/SKILL.md` in its workdir. Claude Code often reads
60
+ the file directly instead of calling its Skill tool, and either way the same
61
+ instructions reach its context.
62
+
63
+ An out-of-lane scenario, where the right answer is to decline the skill, sets
64
+ `"skill_use": "optional"` in `criteria.json`. The assertion then always passes,
65
+ and the result still records whether the skill was loaded. Omitted, it is
66
+ `"required"`. There is no test-level threshold: both must pass. The
58
67
  reported score is the assert-set's weighted score; skill-used is reported
59
68
  separately as a rate across trials.
60
69
 
@@ -62,14 +71,28 @@ Write descriptions a judge can check against the deliverable: an observable
62
71
  property, not a feeling. Weight the items that would make a reviewer reject the
63
72
  work.
64
73
 
74
+ The agent runs online, like a real session: it has a shell and web access on
75
+ every harness (Claude: Bash, WebFetch, WebSearch; Codex: network and web search
76
+ in its `workspace-write` sandbox; Grok: its CLI defaults). It runs on the
77
+ operator's machine with the operator's logins (Codex gets a per-run
78
+ `CODEX_HOME` carrying only its config and login, so the operator's own skills
79
+ and global guidance stay out), so a task must never ask for a
80
+ live mutation such as posting a comment, pushing, publishing, or writing to a
81
+ shared workspace. Put that state in fixture files and grade the plan. Checks
82
+ the agent can run for itself (install, build, test, fetch a public page) are
83
+ fair to require.
84
+
85
+ Name a specific tool or version only when the skill teaches it. Otherwise grade
86
+ the property the tool provides, so an equivalent approach passes.
87
+
65
88
  ## What the judge sees
66
89
 
67
90
  The agent's final message, plus every file in the workdir that differs from the
68
91
  pre-run manifest. Unchanged inputs are omitted; deleted inputs, unreadable
69
92
  files, and non-regular files are named rather than read.
70
93
 
71
- Sections are sorted by path, each file is capped at 4,000 characters and the
72
- appended total at 24,000, with truncation stated inline. Very large outputs make
94
+ Sections are sorted by path, each file is capped at 16,000 characters and the
95
+ appended total at 64,000, with truncation stated inline. Very large outputs make
73
96
  rubric judges return nothing at all, which is why the caps exist. Keep fixtures
74
97
  small enough that the deliverable fits.
75
98
 
@@ -84,8 +107,10 @@ frontmatter block; body text mentioning the key does not count.
84
107
 
85
108
  ## The workdir
86
109
 
87
- Per run, under `<root>/.skillcheck/scratch/<name>/`, rebuilt from scratch each
88
- time. The skill under test is installed where the harness discovers skills:
110
+ Per run, under a scratch directory in the system temp dir
111
+ (`skillcheck-<hash of the root>/<name>/trial-<n>/`), rebuilt from scratch each
112
+ time. It stays outside the root so the agent cannot reach the skill's source,
113
+ its evals, or the repository's own agent guidance. The skill under test is installed where the harness discovers skills:
89
114
  `.claude/skills/<skill>/`, plus `.agents/skills/<skill>/` on codex, or
90
115
  `.grok/skills/<skill>/` on Grok, with its `evals/` directory excluded, so
91
116
  criteria never leak into the agent's context.
package/docs/usage.md CHANGED
@@ -38,7 +38,7 @@ skillcheck run <scenario-dir> --harness grok
38
38
  skillcheck run <scenario-dir> --trials 3 --agent-effort medium
39
39
  ```
40
40
 
41
- Materializes the scenario into `<root>/.skillcheck/scratch/<name>/trial-<n>/workdir`,
41
+ Materializes the scenario into a temp-dir workdir per trial ([workdir](scenarios.md#the-workdir)),
42
42
  installs the skill under test into that workdir, drives the agent, and grades
43
43
  the files it wrote. Exit 0 means pass, 1 means graded fail, and 2 means error.
44
44
  Exit 2 covers missing usable promptfoo output or optional eval peers. The
@@ -82,6 +82,18 @@ So is a result with fewer rows than trials, or a row without its checklist and
82
82
  With `--trials` above 1, `run` and `sweep` print min, spread, pass count,
83
83
  skill-used count, and `NOISY`.
84
84
 
85
+ ### Control
86
+
87
+ `--control` runs a scenario without the skill: nothing is installed in the
88
+ workdir, a hidden skill's explicit invocation is dropped from the task, and the
89
+ `skill-used` assertion never fails. Its result sits beside the skill run
90
+ (`<name>--control.json`) with `variant: "control"` in the sidecar, and `sweep
91
+ --control` covers every scenario. `summarize` pairs each scenario with its
92
+ control and adds two columns per skill: the mean control score, and the lift
93
+ (skill score minus control score over the paired scenarios). A scenario whose
94
+ control passes every trial prints as `NO LIFT`: it passes without the skill, so
95
+ it does not test the skill.
96
+
85
97
  ### Agent effort
86
98
 
87
99
  `--agent-effort low|medium|high|xhigh|max` sets the Claude agent's effort.
@@ -96,8 +108,8 @@ Omitting it leaves Claude Code's default. Only `--harness claude` takes it;
96
108
  workdir with the skill under `.grok/skills/`. It uses native streaming events
97
109
  to count a completed read of that skill's `SKILL.md` as `skill-used` evidence.
98
110
  Grok must be logged in locally or have its supported credentials configured.
99
- The run disables web search and subagents and grants edit permission in the
100
- workdir. `--agent` selects a Grok model ID.
111
+ The run disables subagents and grants edit permission in the workdir; web
112
+ search stays on. `--agent` selects a Grok model ID.
101
113
 
102
114
  `--judge` takes either a bare Claude model (graded through the Anthropic
103
115
  selection in [auth](#auth)) or a provider-qualified promptfoo id, passed
@@ -200,6 +212,7 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
200
212
  "judge_model": "claude-opus-5",
201
213
  "judge_effort": null,
202
214
  "trials": 3,
215
+ "agent_access": "online",
203
216
  "aggregate": { "pass": false, "pass_rate": 0.6667, "score": 0.81, "...": "..." },
204
217
  "ran_at": "<ISO timestamp>",
205
218
  "tool_version": "<skillcheck version>"
@@ -207,7 +220,8 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
207
220
  ```
208
221
 
209
222
  `summarize` reads those sidecars and refuses to mix skills-tree revisions or
210
- run configurations (agent model and effort, judge model and effort, trials) in
223
+ run configurations (agent model and effort, judge model and effort, trials, and
224
+ agent access) in
211
225
  one scorecard, including retained rows from partial reruns, unless
212
226
  `--allow-mixed`. Configurations are compared within a harness, since harnesses
213
227
  differ by design. Sidecars written before run configurations were recorded fall
@@ -218,9 +232,10 @@ becomes `mixed` and per-entry shas remain. A result with no sidecar reduces as
218
232
 
219
233
  ## State
220
234
 
221
- `<root>/.skillcheck/` holds `scratch/` and `results/`, both disposable and safe
222
- to gitignore, and `scorecards/`, which is meant to be committed. Nothing is ever
223
- written inside the installed package.
235
+ `<root>/.skillcheck/` holds `results/`, disposable and safe to gitignore, and
236
+ `scorecards/`, which is meant to be committed. Scratch workdirs live under the
237
+ system temp dir, outside the root. Nothing is ever written inside the installed
238
+ package.
224
239
 
225
240
  ## Auth
226
241
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "1.2.0",
3
+ "version": "1.4.0",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {