@uinaf/skillcheck 1.3.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -50,7 +50,7 @@ function parseArgs(argv) {
50
50
  const v = argv[++i];
51
51
  if (v === void 0 || v.startsWith("--")) throw new Error(`${a} needs a value`);
52
52
  flags.set(a, v);
53
- } else if (a === "--all" || a === "--allow-mixed") flags.set(a, true);
53
+ } else if (a === "--all" || a === "--allow-mixed" || a === "--control") flags.set(a, true);
54
54
  else throw new Error(`unknown flag: ${a}`);
55
55
  }
56
56
  return {
@@ -98,6 +98,7 @@ function runOptions(flags) {
98
98
  judgeModel,
99
99
  judgeEffort,
100
100
  maxTurns: flags.has("--max-turns") ? parsePositiveInt("--max-turns", flags.get("--max-turns")) : void 0,
101
+ control: flags.get("--control") === true,
101
102
  trials: flags.has("--trials") ? parsePositiveInt("--trials", flags.get("--trials")) : 1
102
103
  };
103
104
  }
@@ -107,7 +108,8 @@ function runConfigOf(opts) {
107
108
  agent_effort: opts.agentEffort ?? null,
108
109
  judge_model: opts.judgeModel,
109
110
  judge_effort: opts.judgeEffort ?? null,
110
- trials: opts.trials ?? 1
111
+ trials: opts.trials ?? 1,
112
+ agent_access: "online"
111
113
  };
112
114
  }
113
115
  function configKey(c) {
@@ -116,12 +118,13 @@ function configKey(c) {
116
118
  c.agent_effort ?? null,
117
119
  c.judge_model,
118
120
  c.judge_effort ?? null,
119
- c.trials ?? 1
121
+ c.trials ?? 1,
122
+ c.agent_access ?? "offline"
120
123
  ]);
121
124
  }
122
125
  function describeConfig(c) {
123
126
  const effort = (e) => e ? `@${e}` : "";
124
- return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}`;
127
+ return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}, ${c.agent_access ?? "offline"}`;
125
128
  }
126
129
  function assertUniformConfig(entries, allowMixed) {
127
130
  if (allowMixed) return;
@@ -262,7 +265,8 @@ function runScenario(scenarioDir, opts, root) {
262
265
  const identity = {
263
266
  skill,
264
267
  scenario,
265
- harness: opts.harness
268
+ harness: opts.harness,
269
+ variant: opts.control ? "control" : "skill"
266
270
  };
267
271
  fs.writeFileSync(attemptPath(resultPath), JSON.stringify(identity) + "\n");
268
272
  fs.rmSync(resultPath, { force: true });
@@ -329,7 +333,8 @@ function resultRunConfig(raw, meta, harness) {
329
333
  agent_effort: m.agent_effort ?? null,
330
334
  judge_model: m.judge_model,
331
335
  judge_effort: m.judge_effort ?? null,
332
- trials: m.trials ?? 1
336
+ trials: m.trials ?? 1,
337
+ agent_access: m.agent_access === "online" ? "online" : "offline"
333
338
  };
334
339
  const r = raw;
335
340
  const agent = r?.config?.providers?.[0]?.config;
@@ -341,7 +346,8 @@ function resultRunConfig(raw, meta, harness) {
341
346
  agent_effort: typeof agent?.effort === "string" ? agent.effort : null,
342
347
  judge_model: judgeName(judge),
343
348
  judge_effort: typeof judgeEffort === "string" ? judgeEffort : null,
344
- trials: r?.results?.results?.length ?? 1
349
+ trials: r?.results?.results?.length ?? 1,
350
+ agent_access: "offline"
345
351
  };
346
352
  }
347
353
  function readJson(file) {
@@ -371,7 +377,7 @@ function discoverScenarios(root) {
371
377
  }
372
378
  return found.sort();
373
379
  }
374
- const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N]";
380
+ const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N] [--control]";
375
381
  function cmdRun(argv) {
376
382
  const { positional, flags } = parseArgs(argv);
377
383
  if (positional.length !== 1) fail(`usage: skillcheck run <scenario-dir> ${RUN_FLAGS}`);
@@ -396,7 +402,7 @@ function cmdSweep(argv) {
396
402
  const wanted = configKey(runConfigOf(opts));
397
403
  let passed = 0, failed = 0, errored = 0, skipped = 0;
398
404
  for (const dir of discoverScenarios(root)) {
399
- const name = runNameFor(dir, opts.harness);
405
+ const name = runNameFor(dir, opts.harness, opts.control);
400
406
  const resultPath = path.join(resultsDir, `${name}.json`);
401
407
  if (!all && fs.existsSync(resultPath) && !fs.existsSync(attemptPath(resultPath))) {
402
408
  const raw = readJson(resultPath);
@@ -431,7 +437,8 @@ function resultIdentity(file, dir) {
431
437
  if (identity !== null && typeof identity === "object" && "skill" in identity && typeof identity.skill === "string" && "scenario" in identity && typeof identity.scenario === "string" && "harness" in identity && (identity.harness === "claude" || identity.harness === "codex" || identity.harness === "grok")) return {
432
438
  skill: identity.skill,
433
439
  scenario: identity.scenario,
434
- harness: identity.harness
440
+ harness: identity.harness,
441
+ variant: "variant" in identity && identity.variant === "control" ? "control" : "skill"
435
442
  };
436
443
  } catch {}
437
444
  if (/^~v3~[0-9a-f]{64}$/.test(base)) throw new Error(`missing identity metadata for ${file}`);
@@ -444,13 +451,17 @@ function resultIdentity(file, dir) {
444
451
  return part;
445
452
  }
446
453
  };
447
- const suffix = base.match(/--(codex|grok|cursor)$/);
454
+ const unsuffixed = base.replace(/--control$/, "");
455
+ const variant = unsuffixed !== base && unsuffixed.includes("--") ? "control" : "skill";
456
+ const stem = variant === "control" ? unsuffixed : base;
457
+ const suffix = stem.match(/--(codex|grok|cursor)$/);
448
458
  const harness = suffix === null ? "claude" : suffix[1];
449
- const [skill, ...rest] = base.replace(/--(codex|grok|cursor)$/, "").split("--");
459
+ const [skill, ...rest] = stem.replace(/--(codex|grok|cursor)$/, "").split("--");
450
460
  return {
451
461
  skill: decode(skill),
452
462
  scenario: decode(rest.join("--")),
453
- harness
463
+ harness,
464
+ variant
454
465
  };
455
466
  }
456
467
  function reduceResults(dir, allowMixed) {
@@ -475,7 +486,7 @@ function reduceResults(dir, allowMixed) {
475
486
  const raw = readJson(path.join(dir, f));
476
487
  const base = f.replace(/\.json$/, "");
477
488
  const meta = readJson(path.join(dir, `${base}.meta.json`));
478
- const { skill, scenario, harness } = resultIdentity(f, dir);
489
+ const { skill, scenario, harness, variant } = resultIdentity(f, dir);
479
490
  const config = resultRunConfig(raw, meta, harness);
480
491
  const verdict = classifyResult(raw, config.trials);
481
492
  if ("error" in verdict) {
@@ -487,7 +498,8 @@ function reduceResults(dir, allowMixed) {
487
498
  const key = entryKey({
488
499
  skill,
489
500
  scenario,
490
- harness
501
+ harness,
502
+ variant
491
503
  });
492
504
  gradedAt.set(key, Math.max(gradedAt.get(key) ?? 0, fs.statSync(path.join(dir, f)).mtimeMs));
493
505
  const sha = typeof meta?.skills_tree_sha === "string" ? meta.skills_tree_sha : "unattested";
@@ -497,6 +509,7 @@ function reduceResults(dir, allowMixed) {
497
509
  skill,
498
510
  scenario,
499
511
  harness,
512
+ variant,
500
513
  skills_tree_sha: sha,
501
514
  ...stats,
502
515
  ...config,
@@ -516,7 +529,8 @@ function entryKey(e) {
516
529
  return [
517
530
  e.skill,
518
531
  e.scenario,
519
- e.harness
532
+ e.harness,
533
+ e.variant ?? "skill"
520
534
  ].join("\0");
521
535
  }
522
536
  function mergeScorecard(existing, fresh) {
@@ -573,27 +587,47 @@ function cmdSummarize(argv) {
573
587
  scenarios: merged.entries
574
588
  };
575
589
  fs.writeFileSync(out, JSON.stringify(scorecard, null, 2) + "\n");
576
- console.log(`${out}: ${merged.entries.length} scenario(s), ${merged.entries.filter((e) => e.pass).length} passing, ${skipped.length} skipped file(s)`);
590
+ const scenarios = merged.entries.filter((e) => e.variant !== "control");
591
+ console.log(`${out}: ${scenarios.length} scenario(s), ${scenarios.filter((e) => e.pass).length} passing, ${merged.entries.length - scenarios.length} control(s), ${skipped.length} skipped file(s)`);
577
592
  if (existing.length > 0) console.log(`merged into today's scorecard: ${entries.length} from this run, ${merged.carried} carried over`);
578
593
  console.log(`\n${formatSkillTable(summarizeSkills(merged.entries))}`);
579
- for (const e of merged.entries.filter((x) => x.noisy)) console.log(`NOISY ${e.skill}/${e.scenario} (${e.harness}) ${formatStats(e)}`);
594
+ for (const e of merged.entries.filter((x) => x.noisy)) {
595
+ const tag = e.variant === "control" ? ", control" : "";
596
+ console.log(`NOISY ${e.skill}/${e.scenario} (${e.harness}${tag}) ${formatStats(e)}`);
597
+ }
598
+ for (const s of summarizeSkills(merged.entries)) for (const scenario of s.no_lift) console.log(`NO LIFT ${s.skill}/${scenario} (${s.harness}): passes without the skill`);
580
599
  }
581
600
  function summarizeSkills(entries) {
582
601
  const groups = /* @__PURE__ */ new Map();
602
+ const controls = /* @__PURE__ */ new Map();
583
603
  for (const e of entries) {
584
604
  const key = `${e.skill}\0${e.harness}`;
585
- groups.set(key, [...groups.get(key) ?? [], e]);
605
+ if (e.variant === "control") controls.set(`${key}\0${e.scenario}`, e);
606
+ else groups.set(key, [...groups.get(key) ?? [], e]);
586
607
  }
587
608
  const mean = (xs) => round4(xs.reduce((a, b) => a + b, 0) / xs.length);
588
- return [...groups.values()].map((rows) => ({
589
- skill: rows[0].skill,
590
- harness: rows[0].harness,
591
- scenarios: rows.length,
592
- pass_all: rows.filter((r) => r.pass).length,
593
- pass_rate: mean(rows.map((r) => r.pass_rate ?? (r.pass ? 1 : 0))),
594
- score: mean(rows.map((r) => r.score)),
595
- noisy: rows.filter((r) => r.noisy).map((r) => r.scenario)
596
- }));
609
+ return [...groups.entries()].map(([key, rows]) => {
610
+ const paired = rows.flatMap((r) => {
611
+ const c = controls.get(`${key}\0${r.scenario}`);
612
+ return c === void 0 ? [] : [{
613
+ skill: r,
614
+ control: c
615
+ }];
616
+ });
617
+ const controlScore = paired.length === 0 ? null : mean(paired.map((p) => p.control.score));
618
+ return {
619
+ skill: rows[0].skill,
620
+ harness: rows[0].harness,
621
+ scenarios: rows.length,
622
+ pass_all: rows.filter((r) => r.pass).length,
623
+ pass_rate: mean(rows.map((r) => r.pass_rate ?? (r.pass ? 1 : 0))),
624
+ score: mean(rows.map((r) => r.score)),
625
+ noisy: rows.filter((r) => r.noisy).map((r) => r.scenario),
626
+ control_score: controlScore,
627
+ lift: controlScore === null ? null : round4(mean(paired.map((p) => p.skill.score)) - controlScore),
628
+ no_lift: paired.filter((p) => p.control.pass).map((p) => p.skill.scenario)
629
+ };
630
+ });
597
631
  }
598
632
  function formatSkillTable(rows) {
599
633
  const table = [[
@@ -603,7 +637,9 @@ function formatSkillTable(rows) {
603
637
  "pass^k",
604
638
  "pass rate",
605
639
  "score",
606
- "noisy"
640
+ "noisy",
641
+ "control",
642
+ "lift"
607
643
  ], ...rows.map((r) => [
608
644
  r.skill,
609
645
  r.harness,
@@ -611,7 +647,9 @@ function formatSkillTable(rows) {
611
647
  `${r.pass_all}/${r.scenarios}`,
612
648
  r.pass_rate.toFixed(2),
613
649
  r.score.toFixed(2),
614
- String(r.noisy.length)
650
+ String(r.noisy.length),
651
+ r.control_score === null ? "-" : r.control_score.toFixed(2),
652
+ r.lift === null ? "-" : `${r.lift >= 0 ? "+" : ""}${r.lift.toFixed(2)}`
615
653
  ])];
616
654
  const widths = table[0].map((_, i) => Math.max(...table.map((row) => row[i].length)));
617
655
  return table.map((row) => row.map((c, i) => c.padEnd(widths[i])).join(" ").trimEnd()).join("\n");
@@ -27,7 +27,6 @@ var GrokProvider = class {
27
27
  "streaming-json",
28
28
  "--permission-mode",
29
29
  "acceptEdits",
30
- "--disable-web-search",
31
30
  "--no-subagents",
32
31
  ...model ? ["--model", model] : [],
33
32
  "-p",
package/dist/scenario.js CHANGED
@@ -15,13 +15,14 @@ function loadScenario(scenarioDir) {
15
15
  if (criteria.skill_use !== void 0 && criteria.skill_use !== "required" && criteria.skill_use !== "optional") throw new Error(`skill_use must be "required" or "optional" in ${scenarioDir}`);
16
16
  for (const item of criteria.checklist) if (!(typeof item?.name === "string" && item.name.trim() !== "" && typeof item?.description === "string" && item.description.trim() !== "" && Number.isFinite(item?.max_score) && item.max_score > 0)) throw new Error(`invalid checklist item in ${scenarioDir}: ${JSON.stringify(item)}`);
17
17
  const files = [];
18
- let prompt = taskMd.replace(/^=+ FILE: (.+?) =+\n([\s\S]*?)\n=+ END FILE =+$/gm, (_, name, content) => {
18
+ const task = taskMd.replace(/^=+ FILE: (.+?) =+\n([\s\S]*?)\n=+ END FILE =+$/gm, (_, name, content) => {
19
19
  files.push({
20
20
  name: name.trim(),
21
21
  content: content + "\n"
22
22
  });
23
23
  return `(Input file \`${name.trim()}\` is available in your working directory.)`;
24
24
  });
25
+ let prompt = task;
25
26
  const skillDir = path.resolve(scenarioDir, "../..");
26
27
  if (isHiddenSkill(fs.readFileSync(path.join(skillDir, "SKILL.md"), "utf8"))) prompt = `Use the ${skill} skill for this task.\n\n${prompt}`;
27
28
  return {
@@ -30,6 +31,7 @@ function loadScenario(scenarioDir) {
30
31
  name: `${skill}--${scenario}`,
31
32
  skillDir,
32
33
  prompt,
34
+ task,
33
35
  files,
34
36
  criteria
35
37
  };
@@ -38,13 +40,18 @@ function encodeRunNamePart(part) {
38
40
  if (!part.includes("--") && !part.startsWith("-") && !part.endsWith("-") && !part.startsWith("~v2~")) return part;
39
41
  return `~v2~${encodeURIComponent(part).replaceAll("-", "%2D").replaceAll("~", "%7E")}`;
40
42
  }
41
- function runNameFor(scenarioDir, harness) {
43
+ function runNameFor(scenarioDir, harness, control = false) {
42
44
  const m = path.resolve(scenarioDir).match(/skills\/([^/]+)\/evals\/([^/]+)$/);
43
45
  if (!m) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
44
46
  const name = `${encodeRunNamePart(m[1])}--${encodeRunNamePart(m[2])}`;
45
- const full = harness === "claude" ? name : `${name}--${harness}`;
47
+ const full = `${harness === "claude" ? name : `${name}--${harness}`}${control ? "--control" : ""}`;
46
48
  if (Buffer.byteLength(`${full}.json.attempt`) <= 255) return full;
47
- return `~v3~${createHash("sha256").update(JSON.stringify([
49
+ return `~v3~${createHash("sha256").update(JSON.stringify(control ? [
50
+ m[1],
51
+ m[2],
52
+ harness,
53
+ "control"
54
+ ] : [
48
55
  m[1],
49
56
  m[2],
50
57
  harness
@@ -72,7 +79,7 @@ const RESERVED = /* @__PURE__ */ new Set([
72
79
  ".grok",
73
80
  "node_modules"
74
81
  ]);
75
- function materialize(s, runDir, harness) {
82
+ function materialize(s, runDir, harness, control = false) {
76
83
  const workdir = path.join(runDir, "workdir");
77
84
  fs.rmSync(runDir, {
78
85
  recursive: true,
@@ -96,7 +103,7 @@ function materialize(s, runDir, harness) {
96
103
  fs.mkdirSync(path.dirname(dest), { recursive: true });
97
104
  fs.writeFileSync(dest, content);
98
105
  }
99
- const roots = harness === "codex" ? [".claude", ".agents"] : harness === "grok" ? [".grok"] : [".claude"];
106
+ const roots = control ? [] : harness === "codex" ? [".claude", ".agents"] : harness === "grok" ? [".grok"] : [".claude"];
100
107
  for (const root of roots) fs.cpSync(s.skillDir, path.join(workdir, root, "skills", s.skill), {
101
108
  recursive: true,
102
109
  filter: (src) => path.basename(src) !== "evals"
@@ -141,7 +148,9 @@ function agentProvider(opts, workdir, skill, paths) {
141
148
  skip_git_repo_check: true,
142
149
  enable_streaming: true,
143
150
  sandbox_mode: "workspace-write",
144
- cli_env: { CODEX_HOME: process.env.CODEX_HOME ?? path.join(os.homedir(), ".codex") }
151
+ network_access_enabled: true,
152
+ web_search_enabled: true,
153
+ cli_env: { CODEX_HOME: path.join(workdir, "..", "..", "codex-home") }
145
154
  }
146
155
  };
147
156
  return {
@@ -152,14 +161,17 @@ function agentProvider(opts, workdir, skill, paths) {
152
161
  apiKeyRequired: false,
153
162
  working_dir: workdir,
154
163
  setting_sources: ["project"],
155
- skills: [skill],
164
+ ...opts.control ? {} : { skills: [skill] },
156
165
  permission_mode: "acceptEdits",
157
166
  append_allowed_tools: [
158
167
  "Read",
159
168
  "Write",
160
169
  "Edit",
161
170
  "Glob",
162
- "Grep"
171
+ "Grep",
172
+ "Bash",
173
+ "WebFetch",
174
+ "WebSearch"
163
175
  ],
164
176
  max_turns: opts.maxTurns ?? 50
165
177
  }
@@ -219,7 +231,7 @@ function buildConfig(s, trials, opts, paths) {
219
231
  description: s.criteria.context,
220
232
  providers: [trialLabel(i)],
221
233
  vars: {
222
- task: s.prompt,
234
+ task: opts.control ? s.task : s.prompt,
223
235
  workdir: t.workdir,
224
236
  manifest: t.manifestPath
225
237
  },
@@ -237,7 +249,7 @@ function buildConfig(s, trials, opts, paths) {
237
249
  metric: "skill-used",
238
250
  config: {
239
251
  skill: s.skill,
240
- required: s.criteria.skill_use !== "optional"
252
+ required: !opts.control && s.criteria.skill_use !== "optional"
241
253
  }
242
254
  }]
243
255
  }))
@@ -265,15 +277,31 @@ function requiredEvalPackages(opts, hasAnthropicKey) {
265
277
  if (opts.harness === "codex") pkgs.push("@openai/codex-sdk");
266
278
  return pkgs;
267
279
  }
280
+ function privateCodexHome(dir) {
281
+ const source = process.env.CODEX_HOME ?? path.join(os.homedir(), ".codex");
282
+ fs.rmSync(dir, {
283
+ recursive: true,
284
+ force: true
285
+ });
286
+ fs.mkdirSync(dir, {
287
+ recursive: true,
288
+ mode: 448
289
+ });
290
+ for (const file of ["config.toml", "auth.json"]) {
291
+ const from = path.join(source, file);
292
+ if (fs.existsSync(from)) fs.symlinkSync(from, path.join(dir, file));
293
+ }
294
+ }
268
295
  function generateRun(scenarioDir, opts, paths) {
269
296
  const s = loadScenario(scenarioDir);
270
- const name = runNameFor(scenarioDir, opts.harness);
297
+ const name = runNameFor(scenarioDir, opts.harness, opts.control);
271
298
  const runDir = path.join(paths.scratchDir, name);
272
299
  fs.rmSync(runDir, {
273
300
  recursive: true,
274
301
  force: true
275
302
  });
276
- const trials = Array.from({ length: opts.trials ?? 1 }, (_, i) => materialize(s, path.join(runDir, trialLabel(i)), opts.harness));
303
+ const trials = Array.from({ length: opts.trials ?? 1 }, (_, i) => materialize(s, path.join(runDir, trialLabel(i)), opts.harness, opts.control));
304
+ if (opts.harness === "codex") privateCodexHome(path.join(runDir, "codex-home"));
277
305
  const sdkDir = sdkNodeModulesDir();
278
306
  if (sdkDir !== void 0) {
279
307
  const link = path.join(runDir, "node_modules");
@@ -294,4 +322,4 @@ function generateRun(scenarioDir, opts, paths) {
294
322
  };
295
323
  }
296
324
  //#endregion
297
- export { DEFAULT_CLAUDE_AGENT, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
325
+ export { DEFAULT_CLAUDE_AGENT, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, privateCodexHome, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
package/docs/scenarios.md CHANGED
@@ -71,15 +71,16 @@ Write descriptions a judge can check against the deliverable: an observable
71
71
  property, not a feeling. Weight the items that would make a reviewer reject the
72
72
  work.
73
73
 
74
- On the Claude harness the agent can read, search, and write files in its
75
- workdir, but has no shell, so it cannot install, build, test, or reach the
76
- network. The shell stays off because the agent runs on the operator's machine
77
- with the operator's credentials. Codex runs shell commands inside its
78
- `workspace-write` sandbox, and Grok follows its own CLI permission mode. Write criteria a shell-less agent can meet, so one scenario
79
- scores comparably across harnesses: a checklist item that requires live proof,
80
- such as a frozen lockfile or a verified release, fails on Claude. Grade whether
81
- the deliverable names the checks it could not run and hands them over
82
- precisely.
74
+ The agent runs online, like a real session: it has a shell and web access on
75
+ every harness (Claude: Bash, WebFetch, WebSearch; Codex: network and web search
76
+ in its `workspace-write` sandbox; Grok: its CLI defaults). It runs on the
77
+ operator's machine with the operator's logins (Codex gets a per-run
78
+ `CODEX_HOME` carrying only its config and login, so the operator's own skills
79
+ and global guidance stay out), so a task must never ask for a
80
+ live mutation such as posting a comment, pushing, publishing, or writing to a
81
+ shared workspace. Put that state in fixture files and grade the plan. Checks
82
+ the agent can run for itself (install, build, test, fetch a public page) are
83
+ fair to require.
83
84
 
84
85
  Name a specific tool or version only when the skill teaches it. Otherwise grade
85
86
  the property the tool provides, so an equivalent approach passes.
package/docs/usage.md CHANGED
@@ -82,6 +82,18 @@ So is a result with fewer rows than trials, or a row without its checklist and
82
82
  With `--trials` above 1, `run` and `sweep` print min, spread, pass count,
83
83
  skill-used count, and `NOISY`.
84
84
 
85
+ ### Control
86
+
87
+ `--control` runs a scenario without the skill: nothing is installed in the
88
+ workdir, a hidden skill's explicit invocation is dropped from the task, and the
89
+ `skill-used` assertion never fails. Its result sits beside the skill run
90
+ (`<name>--control.json`) with `variant: "control"` in the sidecar, and `sweep
91
+ --control` covers every scenario. `summarize` pairs each scenario with its
92
+ control and adds two columns per skill: the mean control score, and the lift
93
+ (skill score minus control score over the paired scenarios). A scenario whose
94
+ control passes every trial prints as `NO LIFT`: it passes without the skill, so
95
+ it does not test the skill.
96
+
85
97
  ### Agent effort
86
98
 
87
99
  `--agent-effort low|medium|high|xhigh|max` sets the Claude agent's effort.
@@ -96,8 +108,8 @@ Omitting it leaves Claude Code's default. Only `--harness claude` takes it;
96
108
  workdir with the skill under `.grok/skills/`. It uses native streaming events
97
109
  to count a completed read of that skill's `SKILL.md` as `skill-used` evidence.
98
110
  Grok must be logged in locally or have its supported credentials configured.
99
- The run disables web search and subagents and grants edit permission in the
100
- workdir. `--agent` selects a Grok model ID.
111
+ The run disables subagents and grants edit permission in the workdir; web
112
+ search stays on. `--agent` selects a Grok model ID.
101
113
 
102
114
  `--judge` takes either a bare Claude model (graded through the Anthropic
103
115
  selection in [auth](#auth)) or a provider-qualified promptfoo id, passed
@@ -200,6 +212,7 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
200
212
  "judge_model": "claude-opus-5",
201
213
  "judge_effort": null,
202
214
  "trials": 3,
215
+ "agent_access": "online",
203
216
  "aggregate": { "pass": false, "pass_rate": 0.6667, "score": 0.81, "...": "..." },
204
217
  "ran_at": "<ISO timestamp>",
205
218
  "tool_version": "<skillcheck version>"
@@ -207,7 +220,8 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
207
220
  ```
208
221
 
209
222
  `summarize` reads those sidecars and refuses to mix skills-tree revisions or
210
- run configurations (agent model and effort, judge model and effort, trials) in
223
+ run configurations (agent model and effort, judge model and effort, trials, and
224
+ agent access) in
211
225
  one scorecard, including retained rows from partial reruns, unless
212
226
  `--allow-mixed`. Configurations are compared within a harness, since harnesses
213
227
  differ by design. Sidecars written before run configurations were recorded fall
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "1.3.0",
3
+ "version": "1.4.0",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {