@uinaf/skillcheck 1.3.0 → 1.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -50,7 +50,7 @@ function parseArgs(argv) {
50
50
  const v = argv[++i];
51
51
  if (v === void 0 || v.startsWith("--")) throw new Error(`${a} needs a value`);
52
52
  flags.set(a, v);
53
- } else if (a === "--all" || a === "--allow-mixed") flags.set(a, true);
53
+ } else if (a === "--all" || a === "--allow-mixed" || a === "--control") flags.set(a, true);
54
54
  else throw new Error(`unknown flag: ${a}`);
55
55
  }
56
56
  return {
@@ -98,6 +98,7 @@ function runOptions(flags) {
98
98
  judgeModel,
99
99
  judgeEffort,
100
100
  maxTurns: flags.has("--max-turns") ? parsePositiveInt("--max-turns", flags.get("--max-turns")) : void 0,
101
+ control: flags.get("--control") === true,
101
102
  trials: flags.has("--trials") ? parsePositiveInt("--trials", flags.get("--trials")) : 1
102
103
  };
103
104
  }
@@ -107,7 +108,8 @@ function runConfigOf(opts) {
107
108
  agent_effort: opts.agentEffort ?? null,
108
109
  judge_model: opts.judgeModel,
109
110
  judge_effort: opts.judgeEffort ?? null,
110
- trials: opts.trials ?? 1
111
+ trials: opts.trials ?? 1,
112
+ agent_access: "online"
111
113
  };
112
114
  }
113
115
  function configKey(c) {
@@ -116,12 +118,13 @@ function configKey(c) {
116
118
  c.agent_effort ?? null,
117
119
  c.judge_model,
118
120
  c.judge_effort ?? null,
119
- c.trials ?? 1
121
+ c.trials ?? 1,
122
+ c.agent_access ?? "offline"
120
123
  ]);
121
124
  }
122
125
  function describeConfig(c) {
123
126
  const effort = (e) => e ? `@${e}` : "";
124
- return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}`;
127
+ return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}, ${c.agent_access ?? "offline"}`;
125
128
  }
126
129
  function assertUniformConfig(entries, allowMixed) {
127
130
  if (allowMixed) return;
@@ -262,7 +265,8 @@ function runScenario(scenarioDir, opts, root) {
262
265
  const identity = {
263
266
  skill,
264
267
  scenario,
265
- harness: opts.harness
268
+ harness: opts.harness,
269
+ variant: opts.control ? "control" : "skill"
266
270
  };
267
271
  fs.writeFileSync(attemptPath(resultPath), JSON.stringify(identity) + "\n");
268
272
  fs.rmSync(resultPath, { force: true });
@@ -329,7 +333,8 @@ function resultRunConfig(raw, meta, harness) {
329
333
  agent_effort: m.agent_effort ?? null,
330
334
  judge_model: m.judge_model,
331
335
  judge_effort: m.judge_effort ?? null,
332
- trials: m.trials ?? 1
336
+ trials: m.trials ?? 1,
337
+ agent_access: m.agent_access === "online" ? "online" : "offline"
333
338
  };
334
339
  const r = raw;
335
340
  const agent = r?.config?.providers?.[0]?.config;
@@ -341,7 +346,8 @@ function resultRunConfig(raw, meta, harness) {
341
346
  agent_effort: typeof agent?.effort === "string" ? agent.effort : null,
342
347
  judge_model: judgeName(judge),
343
348
  judge_effort: typeof judgeEffort === "string" ? judgeEffort : null,
344
- trials: r?.results?.results?.length ?? 1
349
+ trials: r?.results?.results?.length ?? 1,
350
+ agent_access: "offline"
345
351
  };
346
352
  }
347
353
  function readJson(file) {
@@ -351,6 +357,17 @@ function readJson(file) {
351
357
  return;
352
358
  }
353
359
  }
360
+ function liveScenarios(root) {
361
+ const cli = path.join(root, "cli");
362
+ if (!(fs.existsSync(path.join(root, "skills")) || fs.existsSync(cli) && fs.readdirSync(cli).some((d) => fs.existsSync(path.join(cli, d, "skills"))))) return void 0;
363
+ return new Set(discoverScenarios(root).map((dir) => {
364
+ const parts = dir.split(path.sep);
365
+ return `${parts.at(-3)}\0${parts.at(-1)}`;
366
+ }));
367
+ }
368
+ function isLive(live, e) {
369
+ return live === void 0 || live.has(`${e.skill}\0${e.scenario}`);
370
+ }
354
371
  function discoverScenarios(root) {
355
372
  const roots = [path.join(root, "skills")];
356
373
  const cliDir = path.join(root, "cli");
@@ -371,7 +388,7 @@ function discoverScenarios(root) {
371
388
  }
372
389
  return found.sort();
373
390
  }
374
- const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N]";
391
+ const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N] [--control]";
375
392
  function cmdRun(argv) {
376
393
  const { positional, flags } = parseArgs(argv);
377
394
  if (positional.length !== 1) fail(`usage: skillcheck run <scenario-dir> ${RUN_FLAGS}`);
@@ -396,7 +413,7 @@ function cmdSweep(argv) {
396
413
  const wanted = configKey(runConfigOf(opts));
397
414
  let passed = 0, failed = 0, errored = 0, skipped = 0;
398
415
  for (const dir of discoverScenarios(root)) {
399
- const name = runNameFor(dir, opts.harness);
416
+ const name = runNameFor(dir, opts.harness, opts.control);
400
417
  const resultPath = path.join(resultsDir, `${name}.json`);
401
418
  if (!all && fs.existsSync(resultPath) && !fs.existsSync(attemptPath(resultPath))) {
402
419
  const raw = readJson(resultPath);
@@ -431,7 +448,8 @@ function resultIdentity(file, dir) {
431
448
  if (identity !== null && typeof identity === "object" && "skill" in identity && typeof identity.skill === "string" && "scenario" in identity && typeof identity.scenario === "string" && "harness" in identity && (identity.harness === "claude" || identity.harness === "codex" || identity.harness === "grok")) return {
432
449
  skill: identity.skill,
433
450
  scenario: identity.scenario,
434
- harness: identity.harness
451
+ harness: identity.harness,
452
+ variant: "variant" in identity && identity.variant === "control" ? "control" : "skill"
435
453
  };
436
454
  } catch {}
437
455
  if (/^~v3~[0-9a-f]{64}$/.test(base)) throw new Error(`missing identity metadata for ${file}`);
@@ -444,24 +462,39 @@ function resultIdentity(file, dir) {
444
462
  return part;
445
463
  }
446
464
  };
447
- const suffix = base.match(/--(codex|grok|cursor)$/);
465
+ const unsuffixed = base.replace(/--control$/, "");
466
+ const variant = unsuffixed !== base && unsuffixed.includes("--") ? "control" : "skill";
467
+ const stem = variant === "control" ? unsuffixed : base;
468
+ const suffix = stem.match(/--(codex|grok|cursor)$/);
448
469
  const harness = suffix === null ? "claude" : suffix[1];
449
- const [skill, ...rest] = base.replace(/--(codex|grok|cursor)$/, "").split("--");
470
+ const [skill, ...rest] = stem.replace(/--(codex|grok|cursor)$/, "").split("--");
450
471
  return {
451
472
  skill: decode(skill),
452
473
  scenario: decode(rest.join("--")),
453
- harness
474
+ harness,
475
+ variant
454
476
  };
455
477
  }
456
- function reduceResults(dir, allowMixed) {
478
+ function reduceResults(dir, allowMixed, live) {
457
479
  const entries = [];
458
480
  const skipped = [];
481
+ const retired = /* @__PURE__ */ new Set();
459
482
  const gradedAt = /* @__PURE__ */ new Map();
460
483
  const shas = /* @__PURE__ */ new Set();
461
484
  const files = fs.readdirSync(dir);
462
485
  const incomplete = new Set(files.filter((f) => f.endsWith(".json.attempt")).map((f) => f.replace(/\.attempt$/, "")));
463
486
  const results = files.filter((f) => f.endsWith(".json") && !f.endsWith(".meta.json"));
464
487
  for (const f of [.../* @__PURE__ */ new Set([...results, ...incomplete])].sort()) {
488
+ let identity;
489
+ try {
490
+ identity = resultIdentity(f, dir);
491
+ } catch {
492
+ identity = void 0;
493
+ }
494
+ if (identity !== void 0 && identity.scenario !== "" && !isLive(live, identity)) {
495
+ retired.add(entryKey(identity));
496
+ continue;
497
+ }
465
498
  if (f.endsWith("--cursor.json")) {
466
499
  console.error(`skipping ${f}: Cursor harness is retired`);
467
500
  skipped.push(f);
@@ -475,7 +508,7 @@ function reduceResults(dir, allowMixed) {
475
508
  const raw = readJson(path.join(dir, f));
476
509
  const base = f.replace(/\.json$/, "");
477
510
  const meta = readJson(path.join(dir, `${base}.meta.json`));
478
- const { skill, scenario, harness } = resultIdentity(f, dir);
511
+ const { skill, scenario, harness, variant } = resultIdentity(f, dir);
479
512
  const config = resultRunConfig(raw, meta, harness);
480
513
  const verdict = classifyResult(raw, config.trials);
481
514
  if ("error" in verdict) {
@@ -487,7 +520,8 @@ function reduceResults(dir, allowMixed) {
487
520
  const key = entryKey({
488
521
  skill,
489
522
  scenario,
490
- harness
523
+ harness,
524
+ variant
491
525
  });
492
526
  gradedAt.set(key, Math.max(gradedAt.get(key) ?? 0, fs.statSync(path.join(dir, f)).mtimeMs));
493
527
  const sha = typeof meta?.skills_tree_sha === "string" ? meta.skills_tree_sha : "unattested";
@@ -497,6 +531,7 @@ function reduceResults(dir, allowMixed) {
497
531
  skill,
498
532
  scenario,
499
533
  harness,
534
+ variant,
500
535
  skills_tree_sha: sha,
501
536
  ...stats,
502
537
  ...config,
@@ -509,14 +544,16 @@ function reduceResults(dir, allowMixed) {
509
544
  treeSha: shas.size === 1 ? [...shas][0] : shas.size === 0 ? "none" : "mixed",
510
545
  entries,
511
546
  skipped,
512
- gradedAt
547
+ gradedAt,
548
+ retired
513
549
  };
514
550
  }
515
551
  function entryKey(e) {
516
552
  return [
517
553
  e.skill,
518
554
  e.scenario,
519
- e.harness
555
+ e.harness,
556
+ e.variant ?? "skill"
520
557
  ].join("\0");
521
558
  }
522
559
  function mergeScorecard(existing, fresh) {
@@ -551,12 +588,17 @@ function readExistingScorecard(out) {
551
588
  function cmdSummarize(argv) {
552
589
  const { positional, flags } = parseArgs(argv);
553
590
  if (positional.length > 0) fail("usage: skillcheck summarize [--root DIR] [--allow-mixed]");
554
- const dirs = stateDirs(resolveRoot(flags));
591
+ const root = resolveRoot(flags);
592
+ const dirs = stateDirs(root);
555
593
  if (!fs.existsSync(dirs.results)) fail(`no results directory at ${dirs.results}; run some evals first`);
556
- const { entries, skipped, gradedAt } = reduceResults(dirs.results, flags.get("--allow-mixed") === true);
594
+ const live = liveScenarios(root);
595
+ const { entries, skipped, gradedAt, retired: retiredResults } = reduceResults(dirs.results, flags.get("--allow-mixed") === true, live);
557
596
  fs.mkdirSync(dirs.scorecards, { recursive: true });
558
597
  const out = path.join(dirs.scorecards, `${(/* @__PURE__ */ new Date()).toISOString().slice(0, 10)}.json`);
559
- const existing = readExistingScorecard(out);
598
+ const previous = readExistingScorecard(out);
599
+ const existing = previous.filter((e) => isLive(live, e));
600
+ for (const e of previous) if (!isLive(live, e)) retiredResults.add(entryKey(e));
601
+ const retired = retiredResults.size;
560
602
  const skippedKeys = new Set(skipped.map((file) => ({
561
603
  key: entryKey(resultIdentity(file, dirs.results)),
562
604
  modifiedAt: fs.statSync(fs.existsSync(path.join(dirs.results, `${file}.attempt`)) ? path.join(dirs.results, `${file}.attempt`) : path.join(dirs.results, file)).mtimeMs
@@ -573,27 +615,48 @@ function cmdSummarize(argv) {
573
615
  scenarios: merged.entries
574
616
  };
575
617
  fs.writeFileSync(out, JSON.stringify(scorecard, null, 2) + "\n");
576
- console.log(`${out}: ${merged.entries.length} scenario(s), ${merged.entries.filter((e) => e.pass).length} passing, ${skipped.length} skipped file(s)`);
618
+ if (retired > 0) console.log(`dropped ${retired} row(s) for scenarios no longer in the tree`);
619
+ const scenarios = merged.entries.filter((e) => e.variant !== "control");
620
+ console.log(`${out}: ${scenarios.length} scenario(s), ${scenarios.filter((e) => e.pass).length} passing, ${merged.entries.length - scenarios.length} control(s), ${skipped.length} skipped file(s)`);
577
621
  if (existing.length > 0) console.log(`merged into today's scorecard: ${entries.length} from this run, ${merged.carried} carried over`);
578
622
  console.log(`\n${formatSkillTable(summarizeSkills(merged.entries))}`);
579
- for (const e of merged.entries.filter((x) => x.noisy)) console.log(`NOISY ${e.skill}/${e.scenario} (${e.harness}) ${formatStats(e)}`);
623
+ for (const e of merged.entries.filter((x) => x.noisy)) {
624
+ const tag = e.variant === "control" ? ", control" : "";
625
+ console.log(`NOISY ${e.skill}/${e.scenario} (${e.harness}${tag}) ${formatStats(e)}`);
626
+ }
627
+ for (const s of summarizeSkills(merged.entries)) for (const scenario of s.no_lift) console.log(`NO LIFT ${s.skill}/${scenario} (${s.harness}): passes without the skill`);
580
628
  }
581
629
  function summarizeSkills(entries) {
582
630
  const groups = /* @__PURE__ */ new Map();
631
+ const controls = /* @__PURE__ */ new Map();
583
632
  for (const e of entries) {
584
633
  const key = `${e.skill}\0${e.harness}`;
585
- groups.set(key, [...groups.get(key) ?? [], e]);
634
+ if (e.variant === "control") controls.set(`${key}\0${e.scenario}`, e);
635
+ else groups.set(key, [...groups.get(key) ?? [], e]);
586
636
  }
587
637
  const mean = (xs) => round4(xs.reduce((a, b) => a + b, 0) / xs.length);
588
- return [...groups.values()].map((rows) => ({
589
- skill: rows[0].skill,
590
- harness: rows[0].harness,
591
- scenarios: rows.length,
592
- pass_all: rows.filter((r) => r.pass).length,
593
- pass_rate: mean(rows.map((r) => r.pass_rate ?? (r.pass ? 1 : 0))),
594
- score: mean(rows.map((r) => r.score)),
595
- noisy: rows.filter((r) => r.noisy).map((r) => r.scenario)
596
- }));
638
+ return [...groups.entries()].map(([key, rows]) => {
639
+ const paired = rows.flatMap((r) => {
640
+ const c = controls.get(`${key}\0${r.scenario}`);
641
+ return c === void 0 ? [] : [{
642
+ skill: r,
643
+ control: c
644
+ }];
645
+ });
646
+ const controlScore = paired.length === 0 ? null : mean(paired.map((p) => p.control.score));
647
+ return {
648
+ skill: rows[0].skill,
649
+ harness: rows[0].harness,
650
+ scenarios: rows.length,
651
+ pass_all: rows.filter((r) => r.pass).length,
652
+ pass_rate: mean(rows.map((r) => r.pass_rate ?? (r.pass ? 1 : 0))),
653
+ score: mean(rows.map((r) => r.score)),
654
+ noisy: rows.filter((r) => r.noisy).map((r) => r.scenario),
655
+ control_score: controlScore,
656
+ lift: controlScore === null ? null : round4(mean(paired.map((p) => p.skill.score)) - controlScore),
657
+ no_lift: paired.filter((p) => p.control.pass).map((p) => p.skill.scenario)
658
+ };
659
+ });
597
660
  }
598
661
  function formatSkillTable(rows) {
599
662
  const table = [[
@@ -603,7 +666,9 @@ function formatSkillTable(rows) {
603
666
  "pass^k",
604
667
  "pass rate",
605
668
  "score",
606
- "noisy"
669
+ "noisy",
670
+ "control",
671
+ "lift"
607
672
  ], ...rows.map((r) => [
608
673
  r.skill,
609
674
  r.harness,
@@ -611,7 +676,9 @@ function formatSkillTable(rows) {
611
676
  `${r.pass_all}/${r.scenarios}`,
612
677
  r.pass_rate.toFixed(2),
613
678
  r.score.toFixed(2),
614
- String(r.noisy.length)
679
+ String(r.noisy.length),
680
+ r.control_score === null ? "-" : r.control_score.toFixed(2),
681
+ r.lift === null ? "-" : `${r.lift >= 0 ? "+" : ""}${r.lift.toFixed(2)}`
615
682
  ])];
616
683
  const widths = table[0].map((_, i) => Math.max(...table.map((row) => row[i].length)));
617
684
  return table.map((row) => row.map((c, i) => c.padEnd(widths[i])).join(" ").trimEnd()).join("\n");
@@ -650,4 +717,4 @@ if (isMainModule()) {
650
717
  }
651
718
  }
652
719
  //#endregion
653
- export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, ensurePrivateDir, formatStats, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
720
+ export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, ensurePrivateDir, formatStats, liveScenarios, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
@@ -27,7 +27,6 @@ var GrokProvider = class {
27
27
  "streaming-json",
28
28
  "--permission-mode",
29
29
  "acceptEdits",
30
- "--disable-web-search",
31
30
  "--no-subagents",
32
31
  ...model ? ["--model", model] : [],
33
32
  "-p",
package/dist/scenario.js CHANGED
@@ -15,13 +15,14 @@ function loadScenario(scenarioDir) {
15
15
  if (criteria.skill_use !== void 0 && criteria.skill_use !== "required" && criteria.skill_use !== "optional") throw new Error(`skill_use must be "required" or "optional" in ${scenarioDir}`);
16
16
  for (const item of criteria.checklist) if (!(typeof item?.name === "string" && item.name.trim() !== "" && typeof item?.description === "string" && item.description.trim() !== "" && Number.isFinite(item?.max_score) && item.max_score > 0)) throw new Error(`invalid checklist item in ${scenarioDir}: ${JSON.stringify(item)}`);
17
17
  const files = [];
18
- let prompt = taskMd.replace(/^=+ FILE: (.+?) =+\n([\s\S]*?)\n=+ END FILE =+$/gm, (_, name, content) => {
18
+ const task = taskMd.replace(/^=+ FILE: (.+?) =+\n([\s\S]*?)\n=+ END FILE =+$/gm, (_, name, content) => {
19
19
  files.push({
20
20
  name: name.trim(),
21
21
  content: content + "\n"
22
22
  });
23
23
  return `(Input file \`${name.trim()}\` is available in your working directory.)`;
24
24
  });
25
+ let prompt = task;
25
26
  const skillDir = path.resolve(scenarioDir, "../..");
26
27
  if (isHiddenSkill(fs.readFileSync(path.join(skillDir, "SKILL.md"), "utf8"))) prompt = `Use the ${skill} skill for this task.\n\n${prompt}`;
27
28
  return {
@@ -30,6 +31,7 @@ function loadScenario(scenarioDir) {
30
31
  name: `${skill}--${scenario}`,
31
32
  skillDir,
32
33
  prompt,
34
+ task,
33
35
  files,
34
36
  criteria
35
37
  };
@@ -38,13 +40,18 @@ function encodeRunNamePart(part) {
38
40
  if (!part.includes("--") && !part.startsWith("-") && !part.endsWith("-") && !part.startsWith("~v2~")) return part;
39
41
  return `~v2~${encodeURIComponent(part).replaceAll("-", "%2D").replaceAll("~", "%7E")}`;
40
42
  }
41
- function runNameFor(scenarioDir, harness) {
43
+ function runNameFor(scenarioDir, harness, control = false) {
42
44
  const m = path.resolve(scenarioDir).match(/skills\/([^/]+)\/evals\/([^/]+)$/);
43
45
  if (!m) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
44
46
  const name = `${encodeRunNamePart(m[1])}--${encodeRunNamePart(m[2])}`;
45
- const full = harness === "claude" ? name : `${name}--${harness}`;
47
+ const full = `${harness === "claude" ? name : `${name}--${harness}`}${control ? "--control" : ""}`;
46
48
  if (Buffer.byteLength(`${full}.json.attempt`) <= 255) return full;
47
- return `~v3~${createHash("sha256").update(JSON.stringify([
49
+ return `~v3~${createHash("sha256").update(JSON.stringify(control ? [
50
+ m[1],
51
+ m[2],
52
+ harness,
53
+ "control"
54
+ ] : [
48
55
  m[1],
49
56
  m[2],
50
57
  harness
@@ -72,7 +79,7 @@ const RESERVED = /* @__PURE__ */ new Set([
72
79
  ".grok",
73
80
  "node_modules"
74
81
  ]);
75
- function materialize(s, runDir, harness) {
82
+ function materialize(s, runDir, harness, control = false) {
76
83
  const workdir = path.join(runDir, "workdir");
77
84
  fs.rmSync(runDir, {
78
85
  recursive: true,
@@ -96,7 +103,7 @@ function materialize(s, runDir, harness) {
96
103
  fs.mkdirSync(path.dirname(dest), { recursive: true });
97
104
  fs.writeFileSync(dest, content);
98
105
  }
99
- const roots = harness === "codex" ? [".claude", ".agents"] : harness === "grok" ? [".grok"] : [".claude"];
106
+ const roots = control ? [] : harness === "codex" ? [".claude", ".agents"] : harness === "grok" ? [".grok"] : [".claude"];
100
107
  for (const root of roots) fs.cpSync(s.skillDir, path.join(workdir, root, "skills", s.skill), {
101
108
  recursive: true,
102
109
  filter: (src) => path.basename(src) !== "evals"
@@ -141,7 +148,9 @@ function agentProvider(opts, workdir, skill, paths) {
141
148
  skip_git_repo_check: true,
142
149
  enable_streaming: true,
143
150
  sandbox_mode: "workspace-write",
144
- cli_env: { CODEX_HOME: process.env.CODEX_HOME ?? path.join(os.homedir(), ".codex") }
151
+ network_access_enabled: true,
152
+ web_search_enabled: true,
153
+ cli_env: { CODEX_HOME: path.join(workdir, "..", "..", "codex-home") }
145
154
  }
146
155
  };
147
156
  return {
@@ -152,14 +161,17 @@ function agentProvider(opts, workdir, skill, paths) {
152
161
  apiKeyRequired: false,
153
162
  working_dir: workdir,
154
163
  setting_sources: ["project"],
155
- skills: [skill],
164
+ ...opts.control ? {} : { skills: [skill] },
156
165
  permission_mode: "acceptEdits",
157
166
  append_allowed_tools: [
158
167
  "Read",
159
168
  "Write",
160
169
  "Edit",
161
170
  "Glob",
162
- "Grep"
171
+ "Grep",
172
+ "Bash",
173
+ "WebFetch",
174
+ "WebSearch"
163
175
  ],
164
176
  max_turns: opts.maxTurns ?? 50
165
177
  }
@@ -219,7 +231,7 @@ function buildConfig(s, trials, opts, paths) {
219
231
  description: s.criteria.context,
220
232
  providers: [trialLabel(i)],
221
233
  vars: {
222
- task: s.prompt,
234
+ task: opts.control ? s.task : s.prompt,
223
235
  workdir: t.workdir,
224
236
  manifest: t.manifestPath
225
237
  },
@@ -237,7 +249,7 @@ function buildConfig(s, trials, opts, paths) {
237
249
  metric: "skill-used",
238
250
  config: {
239
251
  skill: s.skill,
240
- required: s.criteria.skill_use !== "optional"
252
+ required: !opts.control && s.criteria.skill_use !== "optional"
241
253
  }
242
254
  }]
243
255
  }))
@@ -265,15 +277,31 @@ function requiredEvalPackages(opts, hasAnthropicKey) {
265
277
  if (opts.harness === "codex") pkgs.push("@openai/codex-sdk");
266
278
  return pkgs;
267
279
  }
280
+ function privateCodexHome(dir) {
281
+ const source = process.env.CODEX_HOME ?? path.join(os.homedir(), ".codex");
282
+ fs.rmSync(dir, {
283
+ recursive: true,
284
+ force: true
285
+ });
286
+ fs.mkdirSync(dir, {
287
+ recursive: true,
288
+ mode: 448
289
+ });
290
+ for (const file of ["config.toml", "auth.json"]) {
291
+ const from = path.join(source, file);
292
+ if (fs.existsSync(from)) fs.symlinkSync(from, path.join(dir, file));
293
+ }
294
+ }
268
295
  function generateRun(scenarioDir, opts, paths) {
269
296
  const s = loadScenario(scenarioDir);
270
- const name = runNameFor(scenarioDir, opts.harness);
297
+ const name = runNameFor(scenarioDir, opts.harness, opts.control);
271
298
  const runDir = path.join(paths.scratchDir, name);
272
299
  fs.rmSync(runDir, {
273
300
  recursive: true,
274
301
  force: true
275
302
  });
276
- const trials = Array.from({ length: opts.trials ?? 1 }, (_, i) => materialize(s, path.join(runDir, trialLabel(i)), opts.harness));
303
+ const trials = Array.from({ length: opts.trials ?? 1 }, (_, i) => materialize(s, path.join(runDir, trialLabel(i)), opts.harness, opts.control));
304
+ if (opts.harness === "codex") privateCodexHome(path.join(runDir, "codex-home"));
277
305
  const sdkDir = sdkNodeModulesDir();
278
306
  if (sdkDir !== void 0) {
279
307
  const link = path.join(runDir, "node_modules");
@@ -294,4 +322,4 @@ function generateRun(scenarioDir, opts, paths) {
294
322
  };
295
323
  }
296
324
  //#endregion
297
- export { DEFAULT_CLAUDE_AGENT, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
325
+ export { DEFAULT_CLAUDE_AGENT, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, privateCodexHome, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
package/docs/scenarios.md CHANGED
@@ -71,15 +71,16 @@ Write descriptions a judge can check against the deliverable: an observable
71
71
  property, not a feeling. Weight the items that would make a reviewer reject the
72
72
  work.
73
73
 
74
- On the Claude harness the agent can read, search, and write files in its
75
- workdir, but has no shell, so it cannot install, build, test, or reach the
76
- network. The shell stays off because the agent runs on the operator's machine
77
- with the operator's credentials. Codex runs shell commands inside its
78
- `workspace-write` sandbox, and Grok follows its own CLI permission mode. Write criteria a shell-less agent can meet, so one scenario
79
- scores comparably across harnesses: a checklist item that requires live proof,
80
- such as a frozen lockfile or a verified release, fails on Claude. Grade whether
81
- the deliverable names the checks it could not run and hands them over
82
- precisely.
74
+ The agent runs online, like a real session: it has a shell and web access on
75
+ every harness (Claude: Bash, WebFetch, WebSearch; Codex: network and web search
76
+ in its `workspace-write` sandbox; Grok: its CLI defaults). It runs on the
77
+ operator's machine with the operator's logins (Codex gets a per-run
78
+ `CODEX_HOME` carrying only its config and login, so the operator's own skills
79
+ and global guidance stay out), so a task must never ask for a
80
+ live mutation such as posting a comment, pushing, publishing, or writing to a
81
+ shared workspace. Put that state in fixture files and grade the plan. Checks
82
+ the agent can run for itself (install, build, test, fetch a public page) are
83
+ fair to require.
83
84
 
84
85
  Name a specific tool or version only when the skill teaches it. Otherwise grade
85
86
  the property the tool provides, so an equivalent approach passes.
package/docs/usage.md CHANGED
@@ -82,6 +82,18 @@ So is a result with fewer rows than trials, or a row without its checklist and
82
82
  With `--trials` above 1, `run` and `sweep` print min, spread, pass count,
83
83
  skill-used count, and `NOISY`.
84
84
 
85
+ ### Control
86
+
87
+ `--control` runs a scenario without the skill: nothing is installed in the
88
+ workdir, a hidden skill's explicit invocation is dropped from the task, and the
89
+ `skill-used` assertion never fails. Its result sits beside the skill run
90
+ (`<name>--control.json`) with `variant: "control"` in the sidecar, and `sweep
91
+ --control` covers every scenario. `summarize` pairs each scenario with its
92
+ control and adds two columns per skill: the mean control score, and the lift
93
+ (skill score minus control score over the paired scenarios). A scenario whose
94
+ control passes every trial prints as `NO LIFT`: it passes without the skill, so
95
+ it does not test the skill.
96
+
85
97
  ### Agent effort
86
98
 
87
99
  `--agent-effort low|medium|high|xhigh|max` sets the Claude agent's effort.
@@ -96,8 +108,8 @@ Omitting it leaves Claude Code's default. Only `--harness claude` takes it;
96
108
  workdir with the skill under `.grok/skills/`. It uses native streaming events
97
109
  to count a completed read of that skill's `SKILL.md` as `skill-used` evidence.
98
110
  Grok must be logged in locally or have its supported credentials configured.
99
- The run disables web search and subagents and grants edit permission in the
100
- workdir. `--agent` selects a Grok model ID.
111
+ The run disables subagents and grants edit permission in the workdir; web
112
+ search stays on. `--agent` selects a Grok model ID.
101
113
 
102
114
  `--judge` takes either a bare Claude model (graded through the Anthropic
103
115
  selection in [auth](#auth)) or a provider-qualified promptfoo id, passed
@@ -167,6 +179,11 @@ six of twenty-nine scenarios therefore leaves twenty-nine rows in the file, not
167
179
  six. A same-date file that cannot be parsed stops the write instead of being
168
180
  overwritten.
169
181
 
182
+ Rows for scenarios that no longer exist in the tree, whether carried from
183
+ today's scorecard or left in `results/`, are dropped and counted on stdout.
184
+ A root with no skills tree (`skills/` or `cli/*/skills/`) holds results only
185
+ and keeps every row.
186
+
170
187
  Files that are not promptfoo results and ungraded transport errors are skipped
171
188
  with a warning rather than failing the reduction. Graded assertion failures
172
189
  remain scored results. If a skipped file matches an existing scorecard row,
@@ -200,6 +217,7 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
200
217
  "judge_model": "claude-opus-5",
201
218
  "judge_effort": null,
202
219
  "trials": 3,
220
+ "agent_access": "online",
203
221
  "aggregate": { "pass": false, "pass_rate": 0.6667, "score": 0.81, "...": "..." },
204
222
  "ran_at": "<ISO timestamp>",
205
223
  "tool_version": "<skillcheck version>"
@@ -207,7 +225,8 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
207
225
  ```
208
226
 
209
227
  `summarize` reads those sidecars and refuses to mix skills-tree revisions or
210
- run configurations (agent model and effort, judge model and effort, trials) in
228
+ run configurations (agent model and effort, judge model and effort, trials, and
229
+ agent access) in
211
230
  one scorecard, including retained rows from partial reruns, unless
212
231
  `--allow-mixed`. Configurations are compared within a harness, since harnesses
213
232
  differ by design. Sidecars written before run configurations were recorded fall
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "1.3.0",
3
+ "version": "1.4.1",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {