@uinaf/skillcheck 1.3.0 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +68 -30
- package/dist/grok-provider.js +0 -1
- package/dist/scenario.js +42 -14
- package/docs/scenarios.md +10 -9
- package/docs/usage.md +17 -3
- package/package.json +1 -1
package/dist/cli.js
CHANGED
|
@@ -50,7 +50,7 @@ function parseArgs(argv) {
|
|
|
50
50
|
const v = argv[++i];
|
|
51
51
|
if (v === void 0 || v.startsWith("--")) throw new Error(`${a} needs a value`);
|
|
52
52
|
flags.set(a, v);
|
|
53
|
-
} else if (a === "--all" || a === "--allow-mixed") flags.set(a, true);
|
|
53
|
+
} else if (a === "--all" || a === "--allow-mixed" || a === "--control") flags.set(a, true);
|
|
54
54
|
else throw new Error(`unknown flag: ${a}`);
|
|
55
55
|
}
|
|
56
56
|
return {
|
|
@@ -98,6 +98,7 @@ function runOptions(flags) {
|
|
|
98
98
|
judgeModel,
|
|
99
99
|
judgeEffort,
|
|
100
100
|
maxTurns: flags.has("--max-turns") ? parsePositiveInt("--max-turns", flags.get("--max-turns")) : void 0,
|
|
101
|
+
control: flags.get("--control") === true,
|
|
101
102
|
trials: flags.has("--trials") ? parsePositiveInt("--trials", flags.get("--trials")) : 1
|
|
102
103
|
};
|
|
103
104
|
}
|
|
@@ -107,7 +108,8 @@ function runConfigOf(opts) {
|
|
|
107
108
|
agent_effort: opts.agentEffort ?? null,
|
|
108
109
|
judge_model: opts.judgeModel,
|
|
109
110
|
judge_effort: opts.judgeEffort ?? null,
|
|
110
|
-
trials: opts.trials ?? 1
|
|
111
|
+
trials: opts.trials ?? 1,
|
|
112
|
+
agent_access: "online"
|
|
111
113
|
};
|
|
112
114
|
}
|
|
113
115
|
function configKey(c) {
|
|
@@ -116,12 +118,13 @@ function configKey(c) {
|
|
|
116
118
|
c.agent_effort ?? null,
|
|
117
119
|
c.judge_model,
|
|
118
120
|
c.judge_effort ?? null,
|
|
119
|
-
c.trials ?? 1
|
|
121
|
+
c.trials ?? 1,
|
|
122
|
+
c.agent_access ?? "offline"
|
|
120
123
|
]);
|
|
121
124
|
}
|
|
122
125
|
function describeConfig(c) {
|
|
123
126
|
const effort = (e) => e ? `@${e}` : "";
|
|
124
|
-
return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}`;
|
|
127
|
+
return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}, ${c.agent_access ?? "offline"}`;
|
|
125
128
|
}
|
|
126
129
|
function assertUniformConfig(entries, allowMixed) {
|
|
127
130
|
if (allowMixed) return;
|
|
@@ -262,7 +265,8 @@ function runScenario(scenarioDir, opts, root) {
|
|
|
262
265
|
const identity = {
|
|
263
266
|
skill,
|
|
264
267
|
scenario,
|
|
265
|
-
harness: opts.harness
|
|
268
|
+
harness: opts.harness,
|
|
269
|
+
variant: opts.control ? "control" : "skill"
|
|
266
270
|
};
|
|
267
271
|
fs.writeFileSync(attemptPath(resultPath), JSON.stringify(identity) + "\n");
|
|
268
272
|
fs.rmSync(resultPath, { force: true });
|
|
@@ -329,7 +333,8 @@ function resultRunConfig(raw, meta, harness) {
|
|
|
329
333
|
agent_effort: m.agent_effort ?? null,
|
|
330
334
|
judge_model: m.judge_model,
|
|
331
335
|
judge_effort: m.judge_effort ?? null,
|
|
332
|
-
trials: m.trials ?? 1
|
|
336
|
+
trials: m.trials ?? 1,
|
|
337
|
+
agent_access: m.agent_access === "online" ? "online" : "offline"
|
|
333
338
|
};
|
|
334
339
|
const r = raw;
|
|
335
340
|
const agent = r?.config?.providers?.[0]?.config;
|
|
@@ -341,7 +346,8 @@ function resultRunConfig(raw, meta, harness) {
|
|
|
341
346
|
agent_effort: typeof agent?.effort === "string" ? agent.effort : null,
|
|
342
347
|
judge_model: judgeName(judge),
|
|
343
348
|
judge_effort: typeof judgeEffort === "string" ? judgeEffort : null,
|
|
344
|
-
trials: r?.results?.results?.length ?? 1
|
|
349
|
+
trials: r?.results?.results?.length ?? 1,
|
|
350
|
+
agent_access: "offline"
|
|
345
351
|
};
|
|
346
352
|
}
|
|
347
353
|
function readJson(file) {
|
|
@@ -371,7 +377,7 @@ function discoverScenarios(root) {
|
|
|
371
377
|
}
|
|
372
378
|
return found.sort();
|
|
373
379
|
}
|
|
374
|
-
const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N]";
|
|
380
|
+
const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N] [--control]";
|
|
375
381
|
function cmdRun(argv) {
|
|
376
382
|
const { positional, flags } = parseArgs(argv);
|
|
377
383
|
if (positional.length !== 1) fail(`usage: skillcheck run <scenario-dir> ${RUN_FLAGS}`);
|
|
@@ -396,7 +402,7 @@ function cmdSweep(argv) {
|
|
|
396
402
|
const wanted = configKey(runConfigOf(opts));
|
|
397
403
|
let passed = 0, failed = 0, errored = 0, skipped = 0;
|
|
398
404
|
for (const dir of discoverScenarios(root)) {
|
|
399
|
-
const name = runNameFor(dir, opts.harness);
|
|
405
|
+
const name = runNameFor(dir, opts.harness, opts.control);
|
|
400
406
|
const resultPath = path.join(resultsDir, `${name}.json`);
|
|
401
407
|
if (!all && fs.existsSync(resultPath) && !fs.existsSync(attemptPath(resultPath))) {
|
|
402
408
|
const raw = readJson(resultPath);
|
|
@@ -431,7 +437,8 @@ function resultIdentity(file, dir) {
|
|
|
431
437
|
if (identity !== null && typeof identity === "object" && "skill" in identity && typeof identity.skill === "string" && "scenario" in identity && typeof identity.scenario === "string" && "harness" in identity && (identity.harness === "claude" || identity.harness === "codex" || identity.harness === "grok")) return {
|
|
432
438
|
skill: identity.skill,
|
|
433
439
|
scenario: identity.scenario,
|
|
434
|
-
harness: identity.harness
|
|
440
|
+
harness: identity.harness,
|
|
441
|
+
variant: "variant" in identity && identity.variant === "control" ? "control" : "skill"
|
|
435
442
|
};
|
|
436
443
|
} catch {}
|
|
437
444
|
if (/^~v3~[0-9a-f]{64}$/.test(base)) throw new Error(`missing identity metadata for ${file}`);
|
|
@@ -444,13 +451,17 @@ function resultIdentity(file, dir) {
|
|
|
444
451
|
return part;
|
|
445
452
|
}
|
|
446
453
|
};
|
|
447
|
-
const
|
|
454
|
+
const unsuffixed = base.replace(/--control$/, "");
|
|
455
|
+
const variant = unsuffixed !== base && unsuffixed.includes("--") ? "control" : "skill";
|
|
456
|
+
const stem = variant === "control" ? unsuffixed : base;
|
|
457
|
+
const suffix = stem.match(/--(codex|grok|cursor)$/);
|
|
448
458
|
const harness = suffix === null ? "claude" : suffix[1];
|
|
449
|
-
const [skill, ...rest] =
|
|
459
|
+
const [skill, ...rest] = stem.replace(/--(codex|grok|cursor)$/, "").split("--");
|
|
450
460
|
return {
|
|
451
461
|
skill: decode(skill),
|
|
452
462
|
scenario: decode(rest.join("--")),
|
|
453
|
-
harness
|
|
463
|
+
harness,
|
|
464
|
+
variant
|
|
454
465
|
};
|
|
455
466
|
}
|
|
456
467
|
function reduceResults(dir, allowMixed) {
|
|
@@ -475,7 +486,7 @@ function reduceResults(dir, allowMixed) {
|
|
|
475
486
|
const raw = readJson(path.join(dir, f));
|
|
476
487
|
const base = f.replace(/\.json$/, "");
|
|
477
488
|
const meta = readJson(path.join(dir, `${base}.meta.json`));
|
|
478
|
-
const { skill, scenario, harness } = resultIdentity(f, dir);
|
|
489
|
+
const { skill, scenario, harness, variant } = resultIdentity(f, dir);
|
|
479
490
|
const config = resultRunConfig(raw, meta, harness);
|
|
480
491
|
const verdict = classifyResult(raw, config.trials);
|
|
481
492
|
if ("error" in verdict) {
|
|
@@ -487,7 +498,8 @@ function reduceResults(dir, allowMixed) {
|
|
|
487
498
|
const key = entryKey({
|
|
488
499
|
skill,
|
|
489
500
|
scenario,
|
|
490
|
-
harness
|
|
501
|
+
harness,
|
|
502
|
+
variant
|
|
491
503
|
});
|
|
492
504
|
gradedAt.set(key, Math.max(gradedAt.get(key) ?? 0, fs.statSync(path.join(dir, f)).mtimeMs));
|
|
493
505
|
const sha = typeof meta?.skills_tree_sha === "string" ? meta.skills_tree_sha : "unattested";
|
|
@@ -497,6 +509,7 @@ function reduceResults(dir, allowMixed) {
|
|
|
497
509
|
skill,
|
|
498
510
|
scenario,
|
|
499
511
|
harness,
|
|
512
|
+
variant,
|
|
500
513
|
skills_tree_sha: sha,
|
|
501
514
|
...stats,
|
|
502
515
|
...config,
|
|
@@ -516,7 +529,8 @@ function entryKey(e) {
|
|
|
516
529
|
return [
|
|
517
530
|
e.skill,
|
|
518
531
|
e.scenario,
|
|
519
|
-
e.harness
|
|
532
|
+
e.harness,
|
|
533
|
+
e.variant ?? "skill"
|
|
520
534
|
].join("\0");
|
|
521
535
|
}
|
|
522
536
|
function mergeScorecard(existing, fresh) {
|
|
@@ -573,27 +587,47 @@ function cmdSummarize(argv) {
|
|
|
573
587
|
scenarios: merged.entries
|
|
574
588
|
};
|
|
575
589
|
fs.writeFileSync(out, JSON.stringify(scorecard, null, 2) + "\n");
|
|
576
|
-
|
|
590
|
+
const scenarios = merged.entries.filter((e) => e.variant !== "control");
|
|
591
|
+
console.log(`${out}: ${scenarios.length} scenario(s), ${scenarios.filter((e) => e.pass).length} passing, ${merged.entries.length - scenarios.length} control(s), ${skipped.length} skipped file(s)`);
|
|
577
592
|
if (existing.length > 0) console.log(`merged into today's scorecard: ${entries.length} from this run, ${merged.carried} carried over`);
|
|
578
593
|
console.log(`\n${formatSkillTable(summarizeSkills(merged.entries))}`);
|
|
579
|
-
for (const e of merged.entries.filter((x) => x.noisy))
|
|
594
|
+
for (const e of merged.entries.filter((x) => x.noisy)) {
|
|
595
|
+
const tag = e.variant === "control" ? ", control" : "";
|
|
596
|
+
console.log(`NOISY ${e.skill}/${e.scenario} (${e.harness}${tag}) ${formatStats(e)}`);
|
|
597
|
+
}
|
|
598
|
+
for (const s of summarizeSkills(merged.entries)) for (const scenario of s.no_lift) console.log(`NO LIFT ${s.skill}/${scenario} (${s.harness}): passes without the skill`);
|
|
580
599
|
}
|
|
581
600
|
function summarizeSkills(entries) {
|
|
582
601
|
const groups = /* @__PURE__ */ new Map();
|
|
602
|
+
const controls = /* @__PURE__ */ new Map();
|
|
583
603
|
for (const e of entries) {
|
|
584
604
|
const key = `${e.skill}\0${e.harness}`;
|
|
585
|
-
|
|
605
|
+
if (e.variant === "control") controls.set(`${key}\0${e.scenario}`, e);
|
|
606
|
+
else groups.set(key, [...groups.get(key) ?? [], e]);
|
|
586
607
|
}
|
|
587
608
|
const mean = (xs) => round4(xs.reduce((a, b) => a + b, 0) / xs.length);
|
|
588
|
-
return [...groups.
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
609
|
+
return [...groups.entries()].map(([key, rows]) => {
|
|
610
|
+
const paired = rows.flatMap((r) => {
|
|
611
|
+
const c = controls.get(`${key}\0${r.scenario}`);
|
|
612
|
+
return c === void 0 ? [] : [{
|
|
613
|
+
skill: r,
|
|
614
|
+
control: c
|
|
615
|
+
}];
|
|
616
|
+
});
|
|
617
|
+
const controlScore = paired.length === 0 ? null : mean(paired.map((p) => p.control.score));
|
|
618
|
+
return {
|
|
619
|
+
skill: rows[0].skill,
|
|
620
|
+
harness: rows[0].harness,
|
|
621
|
+
scenarios: rows.length,
|
|
622
|
+
pass_all: rows.filter((r) => r.pass).length,
|
|
623
|
+
pass_rate: mean(rows.map((r) => r.pass_rate ?? (r.pass ? 1 : 0))),
|
|
624
|
+
score: mean(rows.map((r) => r.score)),
|
|
625
|
+
noisy: rows.filter((r) => r.noisy).map((r) => r.scenario),
|
|
626
|
+
control_score: controlScore,
|
|
627
|
+
lift: controlScore === null ? null : round4(mean(paired.map((p) => p.skill.score)) - controlScore),
|
|
628
|
+
no_lift: paired.filter((p) => p.control.pass).map((p) => p.skill.scenario)
|
|
629
|
+
};
|
|
630
|
+
});
|
|
597
631
|
}
|
|
598
632
|
function formatSkillTable(rows) {
|
|
599
633
|
const table = [[
|
|
@@ -603,7 +637,9 @@ function formatSkillTable(rows) {
|
|
|
603
637
|
"pass^k",
|
|
604
638
|
"pass rate",
|
|
605
639
|
"score",
|
|
606
|
-
"noisy"
|
|
640
|
+
"noisy",
|
|
641
|
+
"control",
|
|
642
|
+
"lift"
|
|
607
643
|
], ...rows.map((r) => [
|
|
608
644
|
r.skill,
|
|
609
645
|
r.harness,
|
|
@@ -611,7 +647,9 @@ function formatSkillTable(rows) {
|
|
|
611
647
|
`${r.pass_all}/${r.scenarios}`,
|
|
612
648
|
r.pass_rate.toFixed(2),
|
|
613
649
|
r.score.toFixed(2),
|
|
614
|
-
String(r.noisy.length)
|
|
650
|
+
String(r.noisy.length),
|
|
651
|
+
r.control_score === null ? "-" : r.control_score.toFixed(2),
|
|
652
|
+
r.lift === null ? "-" : `${r.lift >= 0 ? "+" : ""}${r.lift.toFixed(2)}`
|
|
615
653
|
])];
|
|
616
654
|
const widths = table[0].map((_, i) => Math.max(...table.map((row) => row[i].length)));
|
|
617
655
|
return table.map((row) => row.map((c, i) => c.padEnd(widths[i])).join(" ").trimEnd()).join("\n");
|
package/dist/grok-provider.js
CHANGED
package/dist/scenario.js
CHANGED
|
@@ -15,13 +15,14 @@ function loadScenario(scenarioDir) {
|
|
|
15
15
|
if (criteria.skill_use !== void 0 && criteria.skill_use !== "required" && criteria.skill_use !== "optional") throw new Error(`skill_use must be "required" or "optional" in ${scenarioDir}`);
|
|
16
16
|
for (const item of criteria.checklist) if (!(typeof item?.name === "string" && item.name.trim() !== "" && typeof item?.description === "string" && item.description.trim() !== "" && Number.isFinite(item?.max_score) && item.max_score > 0)) throw new Error(`invalid checklist item in ${scenarioDir}: ${JSON.stringify(item)}`);
|
|
17
17
|
const files = [];
|
|
18
|
-
|
|
18
|
+
const task = taskMd.replace(/^=+ FILE: (.+?) =+\n([\s\S]*?)\n=+ END FILE =+$/gm, (_, name, content) => {
|
|
19
19
|
files.push({
|
|
20
20
|
name: name.trim(),
|
|
21
21
|
content: content + "\n"
|
|
22
22
|
});
|
|
23
23
|
return `(Input file \`${name.trim()}\` is available in your working directory.)`;
|
|
24
24
|
});
|
|
25
|
+
let prompt = task;
|
|
25
26
|
const skillDir = path.resolve(scenarioDir, "../..");
|
|
26
27
|
if (isHiddenSkill(fs.readFileSync(path.join(skillDir, "SKILL.md"), "utf8"))) prompt = `Use the ${skill} skill for this task.\n\n${prompt}`;
|
|
27
28
|
return {
|
|
@@ -30,6 +31,7 @@ function loadScenario(scenarioDir) {
|
|
|
30
31
|
name: `${skill}--${scenario}`,
|
|
31
32
|
skillDir,
|
|
32
33
|
prompt,
|
|
34
|
+
task,
|
|
33
35
|
files,
|
|
34
36
|
criteria
|
|
35
37
|
};
|
|
@@ -38,13 +40,18 @@ function encodeRunNamePart(part) {
|
|
|
38
40
|
if (!part.includes("--") && !part.startsWith("-") && !part.endsWith("-") && !part.startsWith("~v2~")) return part;
|
|
39
41
|
return `~v2~${encodeURIComponent(part).replaceAll("-", "%2D").replaceAll("~", "%7E")}`;
|
|
40
42
|
}
|
|
41
|
-
function runNameFor(scenarioDir, harness) {
|
|
43
|
+
function runNameFor(scenarioDir, harness, control = false) {
|
|
42
44
|
const m = path.resolve(scenarioDir).match(/skills\/([^/]+)\/evals\/([^/]+)$/);
|
|
43
45
|
if (!m) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
|
|
44
46
|
const name = `${encodeRunNamePart(m[1])}--${encodeRunNamePart(m[2])}`;
|
|
45
|
-
const full = harness === "claude" ? name : `${name}--${harness}`;
|
|
47
|
+
const full = `${harness === "claude" ? name : `${name}--${harness}`}${control ? "--control" : ""}`;
|
|
46
48
|
if (Buffer.byteLength(`${full}.json.attempt`) <= 255) return full;
|
|
47
|
-
return `~v3~${createHash("sha256").update(JSON.stringify([
|
|
49
|
+
return `~v3~${createHash("sha256").update(JSON.stringify(control ? [
|
|
50
|
+
m[1],
|
|
51
|
+
m[2],
|
|
52
|
+
harness,
|
|
53
|
+
"control"
|
|
54
|
+
] : [
|
|
48
55
|
m[1],
|
|
49
56
|
m[2],
|
|
50
57
|
harness
|
|
@@ -72,7 +79,7 @@ const RESERVED = /* @__PURE__ */ new Set([
|
|
|
72
79
|
".grok",
|
|
73
80
|
"node_modules"
|
|
74
81
|
]);
|
|
75
|
-
function materialize(s, runDir, harness) {
|
|
82
|
+
function materialize(s, runDir, harness, control = false) {
|
|
76
83
|
const workdir = path.join(runDir, "workdir");
|
|
77
84
|
fs.rmSync(runDir, {
|
|
78
85
|
recursive: true,
|
|
@@ -96,7 +103,7 @@ function materialize(s, runDir, harness) {
|
|
|
96
103
|
fs.mkdirSync(path.dirname(dest), { recursive: true });
|
|
97
104
|
fs.writeFileSync(dest, content);
|
|
98
105
|
}
|
|
99
|
-
const roots = harness === "codex" ? [".claude", ".agents"] : harness === "grok" ? [".grok"] : [".claude"];
|
|
106
|
+
const roots = control ? [] : harness === "codex" ? [".claude", ".agents"] : harness === "grok" ? [".grok"] : [".claude"];
|
|
100
107
|
for (const root of roots) fs.cpSync(s.skillDir, path.join(workdir, root, "skills", s.skill), {
|
|
101
108
|
recursive: true,
|
|
102
109
|
filter: (src) => path.basename(src) !== "evals"
|
|
@@ -141,7 +148,9 @@ function agentProvider(opts, workdir, skill, paths) {
|
|
|
141
148
|
skip_git_repo_check: true,
|
|
142
149
|
enable_streaming: true,
|
|
143
150
|
sandbox_mode: "workspace-write",
|
|
144
|
-
|
|
151
|
+
network_access_enabled: true,
|
|
152
|
+
web_search_enabled: true,
|
|
153
|
+
cli_env: { CODEX_HOME: path.join(workdir, "..", "..", "codex-home") }
|
|
145
154
|
}
|
|
146
155
|
};
|
|
147
156
|
return {
|
|
@@ -152,14 +161,17 @@ function agentProvider(opts, workdir, skill, paths) {
|
|
|
152
161
|
apiKeyRequired: false,
|
|
153
162
|
working_dir: workdir,
|
|
154
163
|
setting_sources: ["project"],
|
|
155
|
-
skills: [skill],
|
|
164
|
+
...opts.control ? {} : { skills: [skill] },
|
|
156
165
|
permission_mode: "acceptEdits",
|
|
157
166
|
append_allowed_tools: [
|
|
158
167
|
"Read",
|
|
159
168
|
"Write",
|
|
160
169
|
"Edit",
|
|
161
170
|
"Glob",
|
|
162
|
-
"Grep"
|
|
171
|
+
"Grep",
|
|
172
|
+
"Bash",
|
|
173
|
+
"WebFetch",
|
|
174
|
+
"WebSearch"
|
|
163
175
|
],
|
|
164
176
|
max_turns: opts.maxTurns ?? 50
|
|
165
177
|
}
|
|
@@ -219,7 +231,7 @@ function buildConfig(s, trials, opts, paths) {
|
|
|
219
231
|
description: s.criteria.context,
|
|
220
232
|
providers: [trialLabel(i)],
|
|
221
233
|
vars: {
|
|
222
|
-
task: s.prompt,
|
|
234
|
+
task: opts.control ? s.task : s.prompt,
|
|
223
235
|
workdir: t.workdir,
|
|
224
236
|
manifest: t.manifestPath
|
|
225
237
|
},
|
|
@@ -237,7 +249,7 @@ function buildConfig(s, trials, opts, paths) {
|
|
|
237
249
|
metric: "skill-used",
|
|
238
250
|
config: {
|
|
239
251
|
skill: s.skill,
|
|
240
|
-
required: s.criteria.skill_use !== "optional"
|
|
252
|
+
required: !opts.control && s.criteria.skill_use !== "optional"
|
|
241
253
|
}
|
|
242
254
|
}]
|
|
243
255
|
}))
|
|
@@ -265,15 +277,31 @@ function requiredEvalPackages(opts, hasAnthropicKey) {
|
|
|
265
277
|
if (opts.harness === "codex") pkgs.push("@openai/codex-sdk");
|
|
266
278
|
return pkgs;
|
|
267
279
|
}
|
|
280
|
+
function privateCodexHome(dir) {
|
|
281
|
+
const source = process.env.CODEX_HOME ?? path.join(os.homedir(), ".codex");
|
|
282
|
+
fs.rmSync(dir, {
|
|
283
|
+
recursive: true,
|
|
284
|
+
force: true
|
|
285
|
+
});
|
|
286
|
+
fs.mkdirSync(dir, {
|
|
287
|
+
recursive: true,
|
|
288
|
+
mode: 448
|
|
289
|
+
});
|
|
290
|
+
for (const file of ["config.toml", "auth.json"]) {
|
|
291
|
+
const from = path.join(source, file);
|
|
292
|
+
if (fs.existsSync(from)) fs.symlinkSync(from, path.join(dir, file));
|
|
293
|
+
}
|
|
294
|
+
}
|
|
268
295
|
function generateRun(scenarioDir, opts, paths) {
|
|
269
296
|
const s = loadScenario(scenarioDir);
|
|
270
|
-
const name = runNameFor(scenarioDir, opts.harness);
|
|
297
|
+
const name = runNameFor(scenarioDir, opts.harness, opts.control);
|
|
271
298
|
const runDir = path.join(paths.scratchDir, name);
|
|
272
299
|
fs.rmSync(runDir, {
|
|
273
300
|
recursive: true,
|
|
274
301
|
force: true
|
|
275
302
|
});
|
|
276
|
-
const trials = Array.from({ length: opts.trials ?? 1 }, (_, i) => materialize(s, path.join(runDir, trialLabel(i)), opts.harness));
|
|
303
|
+
const trials = Array.from({ length: opts.trials ?? 1 }, (_, i) => materialize(s, path.join(runDir, trialLabel(i)), opts.harness, opts.control));
|
|
304
|
+
if (opts.harness === "codex") privateCodexHome(path.join(runDir, "codex-home"));
|
|
277
305
|
const sdkDir = sdkNodeModulesDir();
|
|
278
306
|
if (sdkDir !== void 0) {
|
|
279
307
|
const link = path.join(runDir, "node_modules");
|
|
@@ -294,4 +322,4 @@ function generateRun(scenarioDir, opts, paths) {
|
|
|
294
322
|
};
|
|
295
323
|
}
|
|
296
324
|
//#endregion
|
|
297
|
-
export { DEFAULT_CLAUDE_AGENT, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
|
|
325
|
+
export { DEFAULT_CLAUDE_AGENT, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, privateCodexHome, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
|
package/docs/scenarios.md
CHANGED
|
@@ -71,15 +71,16 @@ Write descriptions a judge can check against the deliverable: an observable
|
|
|
71
71
|
property, not a feeling. Weight the items that would make a reviewer reject the
|
|
72
72
|
work.
|
|
73
73
|
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
with the operator's
|
|
78
|
-
`
|
|
79
|
-
|
|
80
|
-
such as
|
|
81
|
-
|
|
82
|
-
|
|
74
|
+
The agent runs online, like a real session: it has a shell and web access on
|
|
75
|
+
every harness (Claude: Bash, WebFetch, WebSearch; Codex: network and web search
|
|
76
|
+
in its `workspace-write` sandbox; Grok: its CLI defaults). It runs on the
|
|
77
|
+
operator's machine with the operator's logins (Codex gets a per-run
|
|
78
|
+
`CODEX_HOME` carrying only its config and login, so the operator's own skills
|
|
79
|
+
and global guidance stay out), so a task must never ask for a
|
|
80
|
+
live mutation such as posting a comment, pushing, publishing, or writing to a
|
|
81
|
+
shared workspace. Put that state in fixture files and grade the plan. Checks
|
|
82
|
+
the agent can run for itself (install, build, test, fetch a public page) are
|
|
83
|
+
fair to require.
|
|
83
84
|
|
|
84
85
|
Name a specific tool or version only when the skill teaches it. Otherwise grade
|
|
85
86
|
the property the tool provides, so an equivalent approach passes.
|
package/docs/usage.md
CHANGED
|
@@ -82,6 +82,18 @@ So is a result with fewer rows than trials, or a row without its checklist and
|
|
|
82
82
|
With `--trials` above 1, `run` and `sweep` print min, spread, pass count,
|
|
83
83
|
skill-used count, and `NOISY`.
|
|
84
84
|
|
|
85
|
+
### Control
|
|
86
|
+
|
|
87
|
+
`--control` runs a scenario without the skill: nothing is installed in the
|
|
88
|
+
workdir, a hidden skill's explicit invocation is dropped from the task, and the
|
|
89
|
+
`skill-used` assertion never fails. Its result sits beside the skill run
|
|
90
|
+
(`<name>--control.json`) with `variant: "control"` in the sidecar, and `sweep
|
|
91
|
+
--control` covers every scenario. `summarize` pairs each scenario with its
|
|
92
|
+
control and adds two columns per skill: the mean control score, and the lift
|
|
93
|
+
(skill score minus control score over the paired scenarios). A scenario whose
|
|
94
|
+
control passes every trial prints as `NO LIFT`: it passes without the skill, so
|
|
95
|
+
it does not test the skill.
|
|
96
|
+
|
|
85
97
|
### Agent effort
|
|
86
98
|
|
|
87
99
|
`--agent-effort low|medium|high|xhigh|max` sets the Claude agent's effort.
|
|
@@ -96,8 +108,8 @@ Omitting it leaves Claude Code's default. Only `--harness claude` takes it;
|
|
|
96
108
|
workdir with the skill under `.grok/skills/`. It uses native streaming events
|
|
97
109
|
to count a completed read of that skill's `SKILL.md` as `skill-used` evidence.
|
|
98
110
|
Grok must be logged in locally or have its supported credentials configured.
|
|
99
|
-
The run disables
|
|
100
|
-
|
|
111
|
+
The run disables subagents and grants edit permission in the workdir; web
|
|
112
|
+
search stays on. `--agent` selects a Grok model ID.
|
|
101
113
|
|
|
102
114
|
`--judge` takes either a bare Claude model (graded through the Anthropic
|
|
103
115
|
selection in [auth](#auth)) or a provider-qualified promptfoo id, passed
|
|
@@ -200,6 +212,7 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
|
|
|
200
212
|
"judge_model": "claude-opus-5",
|
|
201
213
|
"judge_effort": null,
|
|
202
214
|
"trials": 3,
|
|
215
|
+
"agent_access": "online",
|
|
203
216
|
"aggregate": { "pass": false, "pass_rate": 0.6667, "score": 0.81, "...": "..." },
|
|
204
217
|
"ran_at": "<ISO timestamp>",
|
|
205
218
|
"tool_version": "<skillcheck version>"
|
|
@@ -207,7 +220,8 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
|
|
|
207
220
|
```
|
|
208
221
|
|
|
209
222
|
`summarize` reads those sidecars and refuses to mix skills-tree revisions or
|
|
210
|
-
run configurations (agent model and effort, judge model and effort, trials
|
|
223
|
+
run configurations (agent model and effort, judge model and effort, trials, and
|
|
224
|
+
agent access) in
|
|
211
225
|
one scorecard, including retained rows from partial reruns, unless
|
|
212
226
|
`--allow-mixed`. Configurations are compared within a harness, since harnesses
|
|
213
227
|
differ by design. Sidecars written before run configurations were recorded fall
|