@uinaf/skillcheck 1.3.0 → 1.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +103 -36
- package/dist/grok-provider.js +0 -1
- package/dist/scenario.js +42 -14
- package/docs/scenarios.md +10 -9
- package/docs/usage.md +22 -3
- package/package.json +1 -1
package/dist/cli.js
CHANGED
|
@@ -50,7 +50,7 @@ function parseArgs(argv) {
|
|
|
50
50
|
const v = argv[++i];
|
|
51
51
|
if (v === void 0 || v.startsWith("--")) throw new Error(`${a} needs a value`);
|
|
52
52
|
flags.set(a, v);
|
|
53
|
-
} else if (a === "--all" || a === "--allow-mixed") flags.set(a, true);
|
|
53
|
+
} else if (a === "--all" || a === "--allow-mixed" || a === "--control") flags.set(a, true);
|
|
54
54
|
else throw new Error(`unknown flag: ${a}`);
|
|
55
55
|
}
|
|
56
56
|
return {
|
|
@@ -98,6 +98,7 @@ function runOptions(flags) {
|
|
|
98
98
|
judgeModel,
|
|
99
99
|
judgeEffort,
|
|
100
100
|
maxTurns: flags.has("--max-turns") ? parsePositiveInt("--max-turns", flags.get("--max-turns")) : void 0,
|
|
101
|
+
control: flags.get("--control") === true,
|
|
101
102
|
trials: flags.has("--trials") ? parsePositiveInt("--trials", flags.get("--trials")) : 1
|
|
102
103
|
};
|
|
103
104
|
}
|
|
@@ -107,7 +108,8 @@ function runConfigOf(opts) {
|
|
|
107
108
|
agent_effort: opts.agentEffort ?? null,
|
|
108
109
|
judge_model: opts.judgeModel,
|
|
109
110
|
judge_effort: opts.judgeEffort ?? null,
|
|
110
|
-
trials: opts.trials ?? 1
|
|
111
|
+
trials: opts.trials ?? 1,
|
|
112
|
+
agent_access: "online"
|
|
111
113
|
};
|
|
112
114
|
}
|
|
113
115
|
function configKey(c) {
|
|
@@ -116,12 +118,13 @@ function configKey(c) {
|
|
|
116
118
|
c.agent_effort ?? null,
|
|
117
119
|
c.judge_model,
|
|
118
120
|
c.judge_effort ?? null,
|
|
119
|
-
c.trials ?? 1
|
|
121
|
+
c.trials ?? 1,
|
|
122
|
+
c.agent_access ?? "offline"
|
|
120
123
|
]);
|
|
121
124
|
}
|
|
122
125
|
function describeConfig(c) {
|
|
123
126
|
const effort = (e) => e ? `@${e}` : "";
|
|
124
|
-
return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}`;
|
|
127
|
+
return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}, ${c.agent_access ?? "offline"}`;
|
|
125
128
|
}
|
|
126
129
|
function assertUniformConfig(entries, allowMixed) {
|
|
127
130
|
if (allowMixed) return;
|
|
@@ -262,7 +265,8 @@ function runScenario(scenarioDir, opts, root) {
|
|
|
262
265
|
const identity = {
|
|
263
266
|
skill,
|
|
264
267
|
scenario,
|
|
265
|
-
harness: opts.harness
|
|
268
|
+
harness: opts.harness,
|
|
269
|
+
variant: opts.control ? "control" : "skill"
|
|
266
270
|
};
|
|
267
271
|
fs.writeFileSync(attemptPath(resultPath), JSON.stringify(identity) + "\n");
|
|
268
272
|
fs.rmSync(resultPath, { force: true });
|
|
@@ -329,7 +333,8 @@ function resultRunConfig(raw, meta, harness) {
|
|
|
329
333
|
agent_effort: m.agent_effort ?? null,
|
|
330
334
|
judge_model: m.judge_model,
|
|
331
335
|
judge_effort: m.judge_effort ?? null,
|
|
332
|
-
trials: m.trials ?? 1
|
|
336
|
+
trials: m.trials ?? 1,
|
|
337
|
+
agent_access: m.agent_access === "online" ? "online" : "offline"
|
|
333
338
|
};
|
|
334
339
|
const r = raw;
|
|
335
340
|
const agent = r?.config?.providers?.[0]?.config;
|
|
@@ -341,7 +346,8 @@ function resultRunConfig(raw, meta, harness) {
|
|
|
341
346
|
agent_effort: typeof agent?.effort === "string" ? agent.effort : null,
|
|
342
347
|
judge_model: judgeName(judge),
|
|
343
348
|
judge_effort: typeof judgeEffort === "string" ? judgeEffort : null,
|
|
344
|
-
trials: r?.results?.results?.length ?? 1
|
|
349
|
+
trials: r?.results?.results?.length ?? 1,
|
|
350
|
+
agent_access: "offline"
|
|
345
351
|
};
|
|
346
352
|
}
|
|
347
353
|
function readJson(file) {
|
|
@@ -351,6 +357,17 @@ function readJson(file) {
|
|
|
351
357
|
return;
|
|
352
358
|
}
|
|
353
359
|
}
|
|
360
|
+
function liveScenarios(root) {
|
|
361
|
+
const cli = path.join(root, "cli");
|
|
362
|
+
if (!(fs.existsSync(path.join(root, "skills")) || fs.existsSync(cli) && fs.readdirSync(cli).some((d) => fs.existsSync(path.join(cli, d, "skills"))))) return void 0;
|
|
363
|
+
return new Set(discoverScenarios(root).map((dir) => {
|
|
364
|
+
const parts = dir.split(path.sep);
|
|
365
|
+
return `${parts.at(-3)}\0${parts.at(-1)}`;
|
|
366
|
+
}));
|
|
367
|
+
}
|
|
368
|
+
function isLive(live, e) {
|
|
369
|
+
return live === void 0 || live.has(`${e.skill}\0${e.scenario}`);
|
|
370
|
+
}
|
|
354
371
|
function discoverScenarios(root) {
|
|
355
372
|
const roots = [path.join(root, "skills")];
|
|
356
373
|
const cliDir = path.join(root, "cli");
|
|
@@ -371,7 +388,7 @@ function discoverScenarios(root) {
|
|
|
371
388
|
}
|
|
372
389
|
return found.sort();
|
|
373
390
|
}
|
|
374
|
-
const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N]";
|
|
391
|
+
const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N] [--control]";
|
|
375
392
|
function cmdRun(argv) {
|
|
376
393
|
const { positional, flags } = parseArgs(argv);
|
|
377
394
|
if (positional.length !== 1) fail(`usage: skillcheck run <scenario-dir> ${RUN_FLAGS}`);
|
|
@@ -396,7 +413,7 @@ function cmdSweep(argv) {
|
|
|
396
413
|
const wanted = configKey(runConfigOf(opts));
|
|
397
414
|
let passed = 0, failed = 0, errored = 0, skipped = 0;
|
|
398
415
|
for (const dir of discoverScenarios(root)) {
|
|
399
|
-
const name = runNameFor(dir, opts.harness);
|
|
416
|
+
const name = runNameFor(dir, opts.harness, opts.control);
|
|
400
417
|
const resultPath = path.join(resultsDir, `${name}.json`);
|
|
401
418
|
if (!all && fs.existsSync(resultPath) && !fs.existsSync(attemptPath(resultPath))) {
|
|
402
419
|
const raw = readJson(resultPath);
|
|
@@ -431,7 +448,8 @@ function resultIdentity(file, dir) {
|
|
|
431
448
|
if (identity !== null && typeof identity === "object" && "skill" in identity && typeof identity.skill === "string" && "scenario" in identity && typeof identity.scenario === "string" && "harness" in identity && (identity.harness === "claude" || identity.harness === "codex" || identity.harness === "grok")) return {
|
|
432
449
|
skill: identity.skill,
|
|
433
450
|
scenario: identity.scenario,
|
|
434
|
-
harness: identity.harness
|
|
451
|
+
harness: identity.harness,
|
|
452
|
+
variant: "variant" in identity && identity.variant === "control" ? "control" : "skill"
|
|
435
453
|
};
|
|
436
454
|
} catch {}
|
|
437
455
|
if (/^~v3~[0-9a-f]{64}$/.test(base)) throw new Error(`missing identity metadata for ${file}`);
|
|
@@ -444,24 +462,39 @@ function resultIdentity(file, dir) {
|
|
|
444
462
|
return part;
|
|
445
463
|
}
|
|
446
464
|
};
|
|
447
|
-
const
|
|
465
|
+
const unsuffixed = base.replace(/--control$/, "");
|
|
466
|
+
const variant = unsuffixed !== base && unsuffixed.includes("--") ? "control" : "skill";
|
|
467
|
+
const stem = variant === "control" ? unsuffixed : base;
|
|
468
|
+
const suffix = stem.match(/--(codex|grok|cursor)$/);
|
|
448
469
|
const harness = suffix === null ? "claude" : suffix[1];
|
|
449
|
-
const [skill, ...rest] =
|
|
470
|
+
const [skill, ...rest] = stem.replace(/--(codex|grok|cursor)$/, "").split("--");
|
|
450
471
|
return {
|
|
451
472
|
skill: decode(skill),
|
|
452
473
|
scenario: decode(rest.join("--")),
|
|
453
|
-
harness
|
|
474
|
+
harness,
|
|
475
|
+
variant
|
|
454
476
|
};
|
|
455
477
|
}
|
|
456
|
-
function reduceResults(dir, allowMixed) {
|
|
478
|
+
function reduceResults(dir, allowMixed, live) {
|
|
457
479
|
const entries = [];
|
|
458
480
|
const skipped = [];
|
|
481
|
+
const retired = /* @__PURE__ */ new Set();
|
|
459
482
|
const gradedAt = /* @__PURE__ */ new Map();
|
|
460
483
|
const shas = /* @__PURE__ */ new Set();
|
|
461
484
|
const files = fs.readdirSync(dir);
|
|
462
485
|
const incomplete = new Set(files.filter((f) => f.endsWith(".json.attempt")).map((f) => f.replace(/\.attempt$/, "")));
|
|
463
486
|
const results = files.filter((f) => f.endsWith(".json") && !f.endsWith(".meta.json"));
|
|
464
487
|
for (const f of [.../* @__PURE__ */ new Set([...results, ...incomplete])].sort()) {
|
|
488
|
+
let identity;
|
|
489
|
+
try {
|
|
490
|
+
identity = resultIdentity(f, dir);
|
|
491
|
+
} catch {
|
|
492
|
+
identity = void 0;
|
|
493
|
+
}
|
|
494
|
+
if (identity !== void 0 && identity.scenario !== "" && !isLive(live, identity)) {
|
|
495
|
+
retired.add(entryKey(identity));
|
|
496
|
+
continue;
|
|
497
|
+
}
|
|
465
498
|
if (f.endsWith("--cursor.json")) {
|
|
466
499
|
console.error(`skipping ${f}: Cursor harness is retired`);
|
|
467
500
|
skipped.push(f);
|
|
@@ -475,7 +508,7 @@ function reduceResults(dir, allowMixed) {
|
|
|
475
508
|
const raw = readJson(path.join(dir, f));
|
|
476
509
|
const base = f.replace(/\.json$/, "");
|
|
477
510
|
const meta = readJson(path.join(dir, `${base}.meta.json`));
|
|
478
|
-
const { skill, scenario, harness } = resultIdentity(f, dir);
|
|
511
|
+
const { skill, scenario, harness, variant } = resultIdentity(f, dir);
|
|
479
512
|
const config = resultRunConfig(raw, meta, harness);
|
|
480
513
|
const verdict = classifyResult(raw, config.trials);
|
|
481
514
|
if ("error" in verdict) {
|
|
@@ -487,7 +520,8 @@ function reduceResults(dir, allowMixed) {
|
|
|
487
520
|
const key = entryKey({
|
|
488
521
|
skill,
|
|
489
522
|
scenario,
|
|
490
|
-
harness
|
|
523
|
+
harness,
|
|
524
|
+
variant
|
|
491
525
|
});
|
|
492
526
|
gradedAt.set(key, Math.max(gradedAt.get(key) ?? 0, fs.statSync(path.join(dir, f)).mtimeMs));
|
|
493
527
|
const sha = typeof meta?.skills_tree_sha === "string" ? meta.skills_tree_sha : "unattested";
|
|
@@ -497,6 +531,7 @@ function reduceResults(dir, allowMixed) {
|
|
|
497
531
|
skill,
|
|
498
532
|
scenario,
|
|
499
533
|
harness,
|
|
534
|
+
variant,
|
|
500
535
|
skills_tree_sha: sha,
|
|
501
536
|
...stats,
|
|
502
537
|
...config,
|
|
@@ -509,14 +544,16 @@ function reduceResults(dir, allowMixed) {
|
|
|
509
544
|
treeSha: shas.size === 1 ? [...shas][0] : shas.size === 0 ? "none" : "mixed",
|
|
510
545
|
entries,
|
|
511
546
|
skipped,
|
|
512
|
-
gradedAt
|
|
547
|
+
gradedAt,
|
|
548
|
+
retired
|
|
513
549
|
};
|
|
514
550
|
}
|
|
515
551
|
function entryKey(e) {
|
|
516
552
|
return [
|
|
517
553
|
e.skill,
|
|
518
554
|
e.scenario,
|
|
519
|
-
e.harness
|
|
555
|
+
e.harness,
|
|
556
|
+
e.variant ?? "skill"
|
|
520
557
|
].join("\0");
|
|
521
558
|
}
|
|
522
559
|
function mergeScorecard(existing, fresh) {
|
|
@@ -551,12 +588,17 @@ function readExistingScorecard(out) {
|
|
|
551
588
|
function cmdSummarize(argv) {
|
|
552
589
|
const { positional, flags } = parseArgs(argv);
|
|
553
590
|
if (positional.length > 0) fail("usage: skillcheck summarize [--root DIR] [--allow-mixed]");
|
|
554
|
-
const
|
|
591
|
+
const root = resolveRoot(flags);
|
|
592
|
+
const dirs = stateDirs(root);
|
|
555
593
|
if (!fs.existsSync(dirs.results)) fail(`no results directory at ${dirs.results}; run some evals first`);
|
|
556
|
-
const
|
|
594
|
+
const live = liveScenarios(root);
|
|
595
|
+
const { entries, skipped, gradedAt, retired: retiredResults } = reduceResults(dirs.results, flags.get("--allow-mixed") === true, live);
|
|
557
596
|
fs.mkdirSync(dirs.scorecards, { recursive: true });
|
|
558
597
|
const out = path.join(dirs.scorecards, `${(/* @__PURE__ */ new Date()).toISOString().slice(0, 10)}.json`);
|
|
559
|
-
const
|
|
598
|
+
const previous = readExistingScorecard(out);
|
|
599
|
+
const existing = previous.filter((e) => isLive(live, e));
|
|
600
|
+
for (const e of previous) if (!isLive(live, e)) retiredResults.add(entryKey(e));
|
|
601
|
+
const retired = retiredResults.size;
|
|
560
602
|
const skippedKeys = new Set(skipped.map((file) => ({
|
|
561
603
|
key: entryKey(resultIdentity(file, dirs.results)),
|
|
562
604
|
modifiedAt: fs.statSync(fs.existsSync(path.join(dirs.results, `${file}.attempt`)) ? path.join(dirs.results, `${file}.attempt`) : path.join(dirs.results, file)).mtimeMs
|
|
@@ -573,27 +615,48 @@ function cmdSummarize(argv) {
|
|
|
573
615
|
scenarios: merged.entries
|
|
574
616
|
};
|
|
575
617
|
fs.writeFileSync(out, JSON.stringify(scorecard, null, 2) + "\n");
|
|
576
|
-
console.log(
|
|
618
|
+
if (retired > 0) console.log(`dropped ${retired} row(s) for scenarios no longer in the tree`);
|
|
619
|
+
const scenarios = merged.entries.filter((e) => e.variant !== "control");
|
|
620
|
+
console.log(`${out}: ${scenarios.length} scenario(s), ${scenarios.filter((e) => e.pass).length} passing, ${merged.entries.length - scenarios.length} control(s), ${skipped.length} skipped file(s)`);
|
|
577
621
|
if (existing.length > 0) console.log(`merged into today's scorecard: ${entries.length} from this run, ${merged.carried} carried over`);
|
|
578
622
|
console.log(`\n${formatSkillTable(summarizeSkills(merged.entries))}`);
|
|
579
|
-
for (const e of merged.entries.filter((x) => x.noisy))
|
|
623
|
+
for (const e of merged.entries.filter((x) => x.noisy)) {
|
|
624
|
+
const tag = e.variant === "control" ? ", control" : "";
|
|
625
|
+
console.log(`NOISY ${e.skill}/${e.scenario} (${e.harness}${tag}) ${formatStats(e)}`);
|
|
626
|
+
}
|
|
627
|
+
for (const s of summarizeSkills(merged.entries)) for (const scenario of s.no_lift) console.log(`NO LIFT ${s.skill}/${scenario} (${s.harness}): passes without the skill`);
|
|
580
628
|
}
|
|
581
629
|
function summarizeSkills(entries) {
|
|
582
630
|
const groups = /* @__PURE__ */ new Map();
|
|
631
|
+
const controls = /* @__PURE__ */ new Map();
|
|
583
632
|
for (const e of entries) {
|
|
584
633
|
const key = `${e.skill}\0${e.harness}`;
|
|
585
|
-
|
|
634
|
+
if (e.variant === "control") controls.set(`${key}\0${e.scenario}`, e);
|
|
635
|
+
else groups.set(key, [...groups.get(key) ?? [], e]);
|
|
586
636
|
}
|
|
587
637
|
const mean = (xs) => round4(xs.reduce((a, b) => a + b, 0) / xs.length);
|
|
588
|
-
return [...groups.
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
638
|
+
return [...groups.entries()].map(([key, rows]) => {
|
|
639
|
+
const paired = rows.flatMap((r) => {
|
|
640
|
+
const c = controls.get(`${key}\0${r.scenario}`);
|
|
641
|
+
return c === void 0 ? [] : [{
|
|
642
|
+
skill: r,
|
|
643
|
+
control: c
|
|
644
|
+
}];
|
|
645
|
+
});
|
|
646
|
+
const controlScore = paired.length === 0 ? null : mean(paired.map((p) => p.control.score));
|
|
647
|
+
return {
|
|
648
|
+
skill: rows[0].skill,
|
|
649
|
+
harness: rows[0].harness,
|
|
650
|
+
scenarios: rows.length,
|
|
651
|
+
pass_all: rows.filter((r) => r.pass).length,
|
|
652
|
+
pass_rate: mean(rows.map((r) => r.pass_rate ?? (r.pass ? 1 : 0))),
|
|
653
|
+
score: mean(rows.map((r) => r.score)),
|
|
654
|
+
noisy: rows.filter((r) => r.noisy).map((r) => r.scenario),
|
|
655
|
+
control_score: controlScore,
|
|
656
|
+
lift: controlScore === null ? null : round4(mean(paired.map((p) => p.skill.score)) - controlScore),
|
|
657
|
+
no_lift: paired.filter((p) => p.control.pass).map((p) => p.skill.scenario)
|
|
658
|
+
};
|
|
659
|
+
});
|
|
597
660
|
}
|
|
598
661
|
function formatSkillTable(rows) {
|
|
599
662
|
const table = [[
|
|
@@ -603,7 +666,9 @@ function formatSkillTable(rows) {
|
|
|
603
666
|
"pass^k",
|
|
604
667
|
"pass rate",
|
|
605
668
|
"score",
|
|
606
|
-
"noisy"
|
|
669
|
+
"noisy",
|
|
670
|
+
"control",
|
|
671
|
+
"lift"
|
|
607
672
|
], ...rows.map((r) => [
|
|
608
673
|
r.skill,
|
|
609
674
|
r.harness,
|
|
@@ -611,7 +676,9 @@ function formatSkillTable(rows) {
|
|
|
611
676
|
`${r.pass_all}/${r.scenarios}`,
|
|
612
677
|
r.pass_rate.toFixed(2),
|
|
613
678
|
r.score.toFixed(2),
|
|
614
|
-
String(r.noisy.length)
|
|
679
|
+
String(r.noisy.length),
|
|
680
|
+
r.control_score === null ? "-" : r.control_score.toFixed(2),
|
|
681
|
+
r.lift === null ? "-" : `${r.lift >= 0 ? "+" : ""}${r.lift.toFixed(2)}`
|
|
615
682
|
])];
|
|
616
683
|
const widths = table[0].map((_, i) => Math.max(...table.map((row) => row[i].length)));
|
|
617
684
|
return table.map((row) => row.map((c, i) => c.padEnd(widths[i])).join(" ").trimEnd()).join("\n");
|
|
@@ -650,4 +717,4 @@ if (isMainModule()) {
|
|
|
650
717
|
}
|
|
651
718
|
}
|
|
652
719
|
//#endregion
|
|
653
|
-
export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, ensurePrivateDir, formatStats, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
|
|
720
|
+
export { NOISY_SPREAD, aggregateTrials, assertUniformConfig, classifyResult, ensurePrivateDir, formatStats, liveScenarios, mergeScorecard, parseArgs, parsePositiveInt, reduceResults, resolveRoot, resultRunConfig, runConfigOf, runOptions, stateDirs, summarizeSkills, toolVersion, treeShaOf };
|
package/dist/grok-provider.js
CHANGED
package/dist/scenario.js
CHANGED
|
@@ -15,13 +15,14 @@ function loadScenario(scenarioDir) {
|
|
|
15
15
|
if (criteria.skill_use !== void 0 && criteria.skill_use !== "required" && criteria.skill_use !== "optional") throw new Error(`skill_use must be "required" or "optional" in ${scenarioDir}`);
|
|
16
16
|
for (const item of criteria.checklist) if (!(typeof item?.name === "string" && item.name.trim() !== "" && typeof item?.description === "string" && item.description.trim() !== "" && Number.isFinite(item?.max_score) && item.max_score > 0)) throw new Error(`invalid checklist item in ${scenarioDir}: ${JSON.stringify(item)}`);
|
|
17
17
|
const files = [];
|
|
18
|
-
|
|
18
|
+
const task = taskMd.replace(/^=+ FILE: (.+?) =+\n([\s\S]*?)\n=+ END FILE =+$/gm, (_, name, content) => {
|
|
19
19
|
files.push({
|
|
20
20
|
name: name.trim(),
|
|
21
21
|
content: content + "\n"
|
|
22
22
|
});
|
|
23
23
|
return `(Input file \`${name.trim()}\` is available in your working directory.)`;
|
|
24
24
|
});
|
|
25
|
+
let prompt = task;
|
|
25
26
|
const skillDir = path.resolve(scenarioDir, "../..");
|
|
26
27
|
if (isHiddenSkill(fs.readFileSync(path.join(skillDir, "SKILL.md"), "utf8"))) prompt = `Use the ${skill} skill for this task.\n\n${prompt}`;
|
|
27
28
|
return {
|
|
@@ -30,6 +31,7 @@ function loadScenario(scenarioDir) {
|
|
|
30
31
|
name: `${skill}--${scenario}`,
|
|
31
32
|
skillDir,
|
|
32
33
|
prompt,
|
|
34
|
+
task,
|
|
33
35
|
files,
|
|
34
36
|
criteria
|
|
35
37
|
};
|
|
@@ -38,13 +40,18 @@ function encodeRunNamePart(part) {
|
|
|
38
40
|
if (!part.includes("--") && !part.startsWith("-") && !part.endsWith("-") && !part.startsWith("~v2~")) return part;
|
|
39
41
|
return `~v2~${encodeURIComponent(part).replaceAll("-", "%2D").replaceAll("~", "%7E")}`;
|
|
40
42
|
}
|
|
41
|
-
function runNameFor(scenarioDir, harness) {
|
|
43
|
+
function runNameFor(scenarioDir, harness, control = false) {
|
|
42
44
|
const m = path.resolve(scenarioDir).match(/skills\/([^/]+)\/evals\/([^/]+)$/);
|
|
43
45
|
if (!m) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
|
|
44
46
|
const name = `${encodeRunNamePart(m[1])}--${encodeRunNamePart(m[2])}`;
|
|
45
|
-
const full = harness === "claude" ? name : `${name}--${harness}`;
|
|
47
|
+
const full = `${harness === "claude" ? name : `${name}--${harness}`}${control ? "--control" : ""}`;
|
|
46
48
|
if (Buffer.byteLength(`${full}.json.attempt`) <= 255) return full;
|
|
47
|
-
return `~v3~${createHash("sha256").update(JSON.stringify([
|
|
49
|
+
return `~v3~${createHash("sha256").update(JSON.stringify(control ? [
|
|
50
|
+
m[1],
|
|
51
|
+
m[2],
|
|
52
|
+
harness,
|
|
53
|
+
"control"
|
|
54
|
+
] : [
|
|
48
55
|
m[1],
|
|
49
56
|
m[2],
|
|
50
57
|
harness
|
|
@@ -72,7 +79,7 @@ const RESERVED = /* @__PURE__ */ new Set([
|
|
|
72
79
|
".grok",
|
|
73
80
|
"node_modules"
|
|
74
81
|
]);
|
|
75
|
-
function materialize(s, runDir, harness) {
|
|
82
|
+
function materialize(s, runDir, harness, control = false) {
|
|
76
83
|
const workdir = path.join(runDir, "workdir");
|
|
77
84
|
fs.rmSync(runDir, {
|
|
78
85
|
recursive: true,
|
|
@@ -96,7 +103,7 @@ function materialize(s, runDir, harness) {
|
|
|
96
103
|
fs.mkdirSync(path.dirname(dest), { recursive: true });
|
|
97
104
|
fs.writeFileSync(dest, content);
|
|
98
105
|
}
|
|
99
|
-
const roots = harness === "codex" ? [".claude", ".agents"] : harness === "grok" ? [".grok"] : [".claude"];
|
|
106
|
+
const roots = control ? [] : harness === "codex" ? [".claude", ".agents"] : harness === "grok" ? [".grok"] : [".claude"];
|
|
100
107
|
for (const root of roots) fs.cpSync(s.skillDir, path.join(workdir, root, "skills", s.skill), {
|
|
101
108
|
recursive: true,
|
|
102
109
|
filter: (src) => path.basename(src) !== "evals"
|
|
@@ -141,7 +148,9 @@ function agentProvider(opts, workdir, skill, paths) {
|
|
|
141
148
|
skip_git_repo_check: true,
|
|
142
149
|
enable_streaming: true,
|
|
143
150
|
sandbox_mode: "workspace-write",
|
|
144
|
-
|
|
151
|
+
network_access_enabled: true,
|
|
152
|
+
web_search_enabled: true,
|
|
153
|
+
cli_env: { CODEX_HOME: path.join(workdir, "..", "..", "codex-home") }
|
|
145
154
|
}
|
|
146
155
|
};
|
|
147
156
|
return {
|
|
@@ -152,14 +161,17 @@ function agentProvider(opts, workdir, skill, paths) {
|
|
|
152
161
|
apiKeyRequired: false,
|
|
153
162
|
working_dir: workdir,
|
|
154
163
|
setting_sources: ["project"],
|
|
155
|
-
skills: [skill],
|
|
164
|
+
...opts.control ? {} : { skills: [skill] },
|
|
156
165
|
permission_mode: "acceptEdits",
|
|
157
166
|
append_allowed_tools: [
|
|
158
167
|
"Read",
|
|
159
168
|
"Write",
|
|
160
169
|
"Edit",
|
|
161
170
|
"Glob",
|
|
162
|
-
"Grep"
|
|
171
|
+
"Grep",
|
|
172
|
+
"Bash",
|
|
173
|
+
"WebFetch",
|
|
174
|
+
"WebSearch"
|
|
163
175
|
],
|
|
164
176
|
max_turns: opts.maxTurns ?? 50
|
|
165
177
|
}
|
|
@@ -219,7 +231,7 @@ function buildConfig(s, trials, opts, paths) {
|
|
|
219
231
|
description: s.criteria.context,
|
|
220
232
|
providers: [trialLabel(i)],
|
|
221
233
|
vars: {
|
|
222
|
-
task: s.prompt,
|
|
234
|
+
task: opts.control ? s.task : s.prompt,
|
|
223
235
|
workdir: t.workdir,
|
|
224
236
|
manifest: t.manifestPath
|
|
225
237
|
},
|
|
@@ -237,7 +249,7 @@ function buildConfig(s, trials, opts, paths) {
|
|
|
237
249
|
metric: "skill-used",
|
|
238
250
|
config: {
|
|
239
251
|
skill: s.skill,
|
|
240
|
-
required: s.criteria.skill_use !== "optional"
|
|
252
|
+
required: !opts.control && s.criteria.skill_use !== "optional"
|
|
241
253
|
}
|
|
242
254
|
}]
|
|
243
255
|
}))
|
|
@@ -265,15 +277,31 @@ function requiredEvalPackages(opts, hasAnthropicKey) {
|
|
|
265
277
|
if (opts.harness === "codex") pkgs.push("@openai/codex-sdk");
|
|
266
278
|
return pkgs;
|
|
267
279
|
}
|
|
280
|
+
function privateCodexHome(dir) {
|
|
281
|
+
const source = process.env.CODEX_HOME ?? path.join(os.homedir(), ".codex");
|
|
282
|
+
fs.rmSync(dir, {
|
|
283
|
+
recursive: true,
|
|
284
|
+
force: true
|
|
285
|
+
});
|
|
286
|
+
fs.mkdirSync(dir, {
|
|
287
|
+
recursive: true,
|
|
288
|
+
mode: 448
|
|
289
|
+
});
|
|
290
|
+
for (const file of ["config.toml", "auth.json"]) {
|
|
291
|
+
const from = path.join(source, file);
|
|
292
|
+
if (fs.existsSync(from)) fs.symlinkSync(from, path.join(dir, file));
|
|
293
|
+
}
|
|
294
|
+
}
|
|
268
295
|
function generateRun(scenarioDir, opts, paths) {
|
|
269
296
|
const s = loadScenario(scenarioDir);
|
|
270
|
-
const name = runNameFor(scenarioDir, opts.harness);
|
|
297
|
+
const name = runNameFor(scenarioDir, opts.harness, opts.control);
|
|
271
298
|
const runDir = path.join(paths.scratchDir, name);
|
|
272
299
|
fs.rmSync(runDir, {
|
|
273
300
|
recursive: true,
|
|
274
301
|
force: true
|
|
275
302
|
});
|
|
276
|
-
const trials = Array.from({ length: opts.trials ?? 1 }, (_, i) => materialize(s, path.join(runDir, trialLabel(i)), opts.harness));
|
|
303
|
+
const trials = Array.from({ length: opts.trials ?? 1 }, (_, i) => materialize(s, path.join(runDir, trialLabel(i)), opts.harness, opts.control));
|
|
304
|
+
if (opts.harness === "codex") privateCodexHome(path.join(runDir, "codex-home"));
|
|
277
305
|
const sdkDir = sdkNodeModulesDir();
|
|
278
306
|
if (sdkDir !== void 0) {
|
|
279
307
|
const link = path.join(runDir, "node_modules");
|
|
@@ -294,4 +322,4 @@ function generateRun(scenarioDir, opts, paths) {
|
|
|
294
322
|
};
|
|
295
323
|
}
|
|
296
324
|
//#endregion
|
|
297
|
-
export { DEFAULT_CLAUDE_AGENT, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
|
|
325
|
+
export { DEFAULT_CLAUDE_AGENT, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, privateCodexHome, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
|
package/docs/scenarios.md
CHANGED
|
@@ -71,15 +71,16 @@ Write descriptions a judge can check against the deliverable: an observable
|
|
|
71
71
|
property, not a feeling. Weight the items that would make a reviewer reject the
|
|
72
72
|
work.
|
|
73
73
|
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
with the operator's
|
|
78
|
-
`
|
|
79
|
-
|
|
80
|
-
such as
|
|
81
|
-
|
|
82
|
-
|
|
74
|
+
The agent runs online, like a real session: it has a shell and web access on
|
|
75
|
+
every harness (Claude: Bash, WebFetch, WebSearch; Codex: network and web search
|
|
76
|
+
in its `workspace-write` sandbox; Grok: its CLI defaults). It runs on the
|
|
77
|
+
operator's machine with the operator's logins (Codex gets a per-run
|
|
78
|
+
`CODEX_HOME` carrying only its config and login, so the operator's own skills
|
|
79
|
+
and global guidance stay out), so a task must never ask for a
|
|
80
|
+
live mutation such as posting a comment, pushing, publishing, or writing to a
|
|
81
|
+
shared workspace. Put that state in fixture files and grade the plan. Checks
|
|
82
|
+
the agent can run for itself (install, build, test, fetch a public page) are
|
|
83
|
+
fair to require.
|
|
83
84
|
|
|
84
85
|
Name a specific tool or version only when the skill teaches it. Otherwise grade
|
|
85
86
|
the property the tool provides, so an equivalent approach passes.
|
package/docs/usage.md
CHANGED
|
@@ -82,6 +82,18 @@ So is a result with fewer rows than trials, or a row without its checklist and
|
|
|
82
82
|
With `--trials` above 1, `run` and `sweep` print min, spread, pass count,
|
|
83
83
|
skill-used count, and `NOISY`.
|
|
84
84
|
|
|
85
|
+
### Control
|
|
86
|
+
|
|
87
|
+
`--control` runs a scenario without the skill: nothing is installed in the
|
|
88
|
+
workdir, a hidden skill's explicit invocation is dropped from the task, and the
|
|
89
|
+
`skill-used` assertion never fails. Its result sits beside the skill run
|
|
90
|
+
(`<name>--control.json`) with `variant: "control"` in the sidecar, and `sweep
|
|
91
|
+
--control` covers every scenario. `summarize` pairs each scenario with its
|
|
92
|
+
control and adds two columns per skill: the mean control score, and the lift
|
|
93
|
+
(skill score minus control score over the paired scenarios). A scenario whose
|
|
94
|
+
control passes every trial prints as `NO LIFT`: it passes without the skill, so
|
|
95
|
+
it does not test the skill.
|
|
96
|
+
|
|
85
97
|
### Agent effort
|
|
86
98
|
|
|
87
99
|
`--agent-effort low|medium|high|xhigh|max` sets the Claude agent's effort.
|
|
@@ -96,8 +108,8 @@ Omitting it leaves Claude Code's default. Only `--harness claude` takes it;
|
|
|
96
108
|
workdir with the skill under `.grok/skills/`. It uses native streaming events
|
|
97
109
|
to count a completed read of that skill's `SKILL.md` as `skill-used` evidence.
|
|
98
110
|
Grok must be logged in locally or have its supported credentials configured.
|
|
99
|
-
The run disables
|
|
100
|
-
|
|
111
|
+
The run disables subagents and grants edit permission in the workdir; web
|
|
112
|
+
search stays on. `--agent` selects a Grok model ID.
|
|
101
113
|
|
|
102
114
|
`--judge` takes either a bare Claude model (graded through the Anthropic
|
|
103
115
|
selection in [auth](#auth)) or a provider-qualified promptfoo id, passed
|
|
@@ -167,6 +179,11 @@ six of twenty-nine scenarios therefore leaves twenty-nine rows in the file, not
|
|
|
167
179
|
six. A same-date file that cannot be parsed stops the write instead of being
|
|
168
180
|
overwritten.
|
|
169
181
|
|
|
182
|
+
Rows for scenarios that no longer exist in the tree, whether carried from
|
|
183
|
+
today's scorecard or left in `results/`, are dropped and counted on stdout.
|
|
184
|
+
A root with no skills tree (`skills/` or `cli/*/skills/`) holds results only
|
|
185
|
+
and keeps every row.
|
|
186
|
+
|
|
170
187
|
Files that are not promptfoo results and ungraded transport errors are skipped
|
|
171
188
|
with a warning rather than failing the reduction. Graded assertion failures
|
|
172
189
|
remain scored results. If a skipped file matches an existing scorecard row,
|
|
@@ -200,6 +217,7 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
|
|
|
200
217
|
"judge_model": "claude-opus-5",
|
|
201
218
|
"judge_effort": null,
|
|
202
219
|
"trials": 3,
|
|
220
|
+
"agent_access": "online",
|
|
203
221
|
"aggregate": { "pass": false, "pass_rate": 0.6667, "score": 0.81, "...": "..." },
|
|
204
222
|
"ran_at": "<ISO timestamp>",
|
|
205
223
|
"tool_version": "<skillcheck version>"
|
|
@@ -207,7 +225,8 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
|
|
|
207
225
|
```
|
|
208
226
|
|
|
209
227
|
`summarize` reads those sidecars and refuses to mix skills-tree revisions or
|
|
210
|
-
run configurations (agent model and effort, judge model and effort, trials
|
|
228
|
+
run configurations (agent model and effort, judge model and effort, trials, and
|
|
229
|
+
agent access) in
|
|
211
230
|
one scorecard, including retained rows from partial reruns, unless
|
|
212
231
|
`--allow-mixed`. Configurations are compared within a harness, since harnesses
|
|
213
232
|
differ by design. Sidecars written before run configurations were recorded fall
|