@uinaf/skillcheck 1.5.3 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +16 -6
- package/dist/scenario.js +27 -4
- package/docs/usage.md +42 -2
- package/package.json +1 -1
package/dist/cli.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import { encodeRunNamePart, generateRun, requiredEvalPackages, resolvePackageDir, runNameFor } from "./scenario.js";
|
|
2
|
+
import { encodeRunNamePart, generateRun, isIsolated, requiredEvalPackages, resolvePackageDir, runNameFor } from "./scenario.js";
|
|
3
3
|
import { lintSkills } from "./lint.js";
|
|
4
4
|
import { execFileSync, spawnSync } from "node:child_process";
|
|
5
5
|
import { createHash } from "node:crypto";
|
|
@@ -110,7 +110,8 @@ function runConfigOf(opts) {
|
|
|
110
110
|
judge_model: opts.judgeModel,
|
|
111
111
|
judge_effort: opts.judgeEffort ?? null,
|
|
112
112
|
trials: opts.trials ?? 1,
|
|
113
|
-
agent_access: "online"
|
|
113
|
+
agent_access: "online",
|
|
114
|
+
isolated: isIsolated()
|
|
114
115
|
};
|
|
115
116
|
}
|
|
116
117
|
function configKey(c) {
|
|
@@ -120,12 +121,13 @@ function configKey(c) {
|
|
|
120
121
|
c.judge_model,
|
|
121
122
|
c.judge_effort ?? null,
|
|
122
123
|
c.trials ?? 1,
|
|
123
|
-
c.agent_access ?? "offline"
|
|
124
|
+
c.agent_access ?? "offline",
|
|
125
|
+
c.isolated ?? false
|
|
124
126
|
]);
|
|
125
127
|
}
|
|
126
128
|
function describeConfig(c) {
|
|
127
129
|
const effort = (e) => e ? `@${e}` : "";
|
|
128
|
-
return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}, ${c.agent_access ?? "offline"}`;
|
|
130
|
+
return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}, ${c.agent_access ?? "offline"}, ${c.isolated ? "isolated" : "host"}`;
|
|
129
131
|
}
|
|
130
132
|
function assertUniformConfig(entries, allowMixed) {
|
|
131
133
|
if (allowMixed) return;
|
|
@@ -335,7 +337,8 @@ function resultRunConfig(raw, meta, harness) {
|
|
|
335
337
|
judge_model: m.judge_model,
|
|
336
338
|
judge_effort: m.judge_effort ?? null,
|
|
337
339
|
trials: m.trials ?? 1,
|
|
338
|
-
agent_access: m.agent_access === "online" ? "online" : "offline"
|
|
340
|
+
agent_access: m.agent_access === "online" ? "online" : "offline",
|
|
341
|
+
isolated: m.isolated === true
|
|
339
342
|
};
|
|
340
343
|
const r = raw;
|
|
341
344
|
const agent = r?.config?.providers?.[0]?.config;
|
|
@@ -348,7 +351,8 @@ function resultRunConfig(raw, meta, harness) {
|
|
|
348
351
|
judge_model: judgeName(judge),
|
|
349
352
|
judge_effort: typeof judgeEffort === "string" ? judgeEffort : null,
|
|
350
353
|
trials: r?.results?.results?.length ?? 1,
|
|
351
|
-
agent_access: "offline"
|
|
354
|
+
agent_access: "offline",
|
|
355
|
+
isolated: false
|
|
352
356
|
};
|
|
353
357
|
}
|
|
354
358
|
function readJson(file) {
|
|
@@ -390,11 +394,16 @@ function discoverScenarios(root) {
|
|
|
390
394
|
return found.sort();
|
|
391
395
|
}
|
|
392
396
|
const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N] [--control]";
|
|
397
|
+
function warnHostRun() {
|
|
398
|
+
if (isIsolated()) return;
|
|
399
|
+
console.error("warning: host run; the agent can read this machine's installed skills, plugins, and files, so scores are not isolated. Run in the isolated image: docs/usage.md#isolated-runs");
|
|
400
|
+
}
|
|
393
401
|
function cmdRun(argv) {
|
|
394
402
|
const { positional, flags } = parseArgs(argv);
|
|
395
403
|
if (positional.length !== 1) fail(`usage: skillcheck run <scenario-dir> ${RUN_FLAGS}`);
|
|
396
404
|
const opts = runOptions(flags);
|
|
397
405
|
ensureEvalPackages(opts);
|
|
406
|
+
warnHostRun();
|
|
398
407
|
const o = runScenario(positional[0], opts, resolveRoot(flags));
|
|
399
408
|
if (o.stats === void 0) {
|
|
400
409
|
console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
|
|
@@ -409,6 +418,7 @@ function cmdSweep(argv) {
|
|
|
409
418
|
const root = resolveRoot(flags);
|
|
410
419
|
const opts = runOptions(flags);
|
|
411
420
|
ensureEvalPackages(opts);
|
|
421
|
+
warnHostRun();
|
|
412
422
|
const all = flags.get("--all") === true;
|
|
413
423
|
const resultsDir = stateDirs(root).results;
|
|
414
424
|
const wanted = configKey(runConfigOf(opts));
|
package/dist/scenario.js
CHANGED
|
@@ -75,6 +75,10 @@ function stripHiddenFlag(skillMd) {
|
|
|
75
75
|
if (!range) return skillMd;
|
|
76
76
|
return skillMd.split("\n").filter((l, i) => !(i >= range[0] && i < range[1] && l.startsWith("disable-model-invocation:"))).join("\n");
|
|
77
77
|
}
|
|
78
|
+
function isIsolated() {
|
|
79
|
+
return process.env.SKILLCHECK_ISOLATED === "1" && (fs.existsSync("/.dockerenv") || fs.existsSync("/run/.containerenv")) && process.getuid?.() === 0 && Number.isInteger(Number(process.env.SKILLCHECK_AGENT_UID)) && fs.existsSync(AGENT_WRAPPER);
|
|
80
|
+
}
|
|
81
|
+
const AGENT_WRAPPER = "/usr/local/libexec/skillcheck/agent-wrapper.sh";
|
|
78
82
|
const MARKDOWN_LINK = /\]\(\s*<([^>]+)>|\]\(\s*([^)\s]+)|^ {0,3}\[[^\]]+\]:\s*<?([^\s>]+)/gm;
|
|
79
83
|
function linkedSiblings(skillDir) {
|
|
80
84
|
const skillsRoot = path.dirname(skillDir);
|
|
@@ -167,7 +171,7 @@ function materialize(s, runDir, harness, control = false) {
|
|
|
167
171
|
};
|
|
168
172
|
walk(workdir);
|
|
169
173
|
const manifestPath = path.join(runDir, "manifest.json");
|
|
170
|
-
fs.writeFileSync(manifestPath, JSON.stringify(manifest, null, 2));
|
|
174
|
+
fs.writeFileSync(manifestPath, JSON.stringify(manifest, null, 2), { mode: 384 });
|
|
171
175
|
return {
|
|
172
176
|
workdir,
|
|
173
177
|
manifestPath
|
|
@@ -189,7 +193,7 @@ function agentProvider(opts, workdir, skill, paths) {
|
|
|
189
193
|
working_dir: workdir,
|
|
190
194
|
skip_git_repo_check: true,
|
|
191
195
|
enable_streaming: true,
|
|
192
|
-
sandbox_mode: "workspace-write",
|
|
196
|
+
sandbox_mode: isIsolated() ? "danger-full-access" : "workspace-write",
|
|
193
197
|
network_access_enabled: true,
|
|
194
198
|
web_search_enabled: true,
|
|
195
199
|
cli_config: { features: { plugins: false } },
|
|
@@ -358,6 +362,24 @@ function privateHome(dir) {
|
|
|
358
362
|
fs.symlinkSync(path.join(source, name), path.join(dir, name));
|
|
359
363
|
}
|
|
360
364
|
}
|
|
365
|
+
function handOverToAgent(runDir, trials) {
|
|
366
|
+
const uid = Number(process.env.SKILLCHECK_AGENT_UID);
|
|
367
|
+
const gid = Number(process.env.SKILLCHECK_AGENT_GID ?? process.env.SKILLCHECK_AGENT_UID);
|
|
368
|
+
if (!Number.isInteger(uid) || !Number.isInteger(gid)) throw new Error("isolated run without SKILLCHECK_AGENT_UID: refusing to run the agent as root");
|
|
369
|
+
const own = (p) => {
|
|
370
|
+
fs.lchownSync(p, uid, gid);
|
|
371
|
+
if (fs.lstatSync(p).isDirectory()) for (const e of fs.readdirSync(p)) own(path.join(p, e));
|
|
372
|
+
};
|
|
373
|
+
for (const dir of [path.dirname(runDir), runDir]) fs.chmodSync(dir, 457);
|
|
374
|
+
for (const t of trials) {
|
|
375
|
+
fs.chmodSync(path.dirname(t.workdir), 457);
|
|
376
|
+
own(t.workdir);
|
|
377
|
+
}
|
|
378
|
+
for (const home of ["codex-home", "home"]) {
|
|
379
|
+
const p = path.join(runDir, home);
|
|
380
|
+
if (fs.existsSync(p)) own(p);
|
|
381
|
+
}
|
|
382
|
+
}
|
|
361
383
|
function generateRun(scenarioDir, opts, paths) {
|
|
362
384
|
const s = loadScenario(scenarioDir);
|
|
363
385
|
const name = runNameFor(scenarioDir, opts.harness, opts.control);
|
|
@@ -382,7 +404,8 @@ function generateRun(scenarioDir, opts, paths) {
|
|
|
382
404
|
}
|
|
383
405
|
const config = buildConfig(s, trials, opts, paths);
|
|
384
406
|
const configPath = path.join(runDir, "promptfooconfig.json");
|
|
385
|
-
fs.writeFileSync(configPath, JSON.stringify(config, null, 2));
|
|
407
|
+
fs.writeFileSync(configPath, JSON.stringify(config, null, 2), { mode: 384 });
|
|
408
|
+
if (isIsolated()) handOverToAgent(runDir, trials);
|
|
386
409
|
return {
|
|
387
410
|
name,
|
|
388
411
|
configPath,
|
|
@@ -391,4 +414,4 @@ function generateRun(scenarioDir, opts, paths) {
|
|
|
391
414
|
};
|
|
392
415
|
}
|
|
393
416
|
//#endregion
|
|
394
|
-
export { DEFAULT_CLAUDE_AGENT, FILE_BLOCK, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, linkedSiblings, loadScenario, materialize, privateCodexHome, privateHome, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
|
|
417
|
+
export { DEFAULT_CLAUDE_AGENT, FILE_BLOCK, buildConfig, encodeRunNamePart, generateRun, handOverToAgent, isHiddenSkill, isIsolated, linkedSiblings, loadScenario, materialize, privateCodexHome, privateHome, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
|
package/docs/usage.md
CHANGED
|
@@ -57,6 +57,45 @@ and a Claude agent limit of 50 turns. `--max-turns` changes that limit only for
|
|
|
57
57
|
Claude; passing it with `codex` or `grok` fails before the eval starts. On
|
|
58
58
|
those harnesses, omitting `--agent` leaves the model to that CLI's own default.
|
|
59
59
|
|
|
60
|
+
### Isolated runs
|
|
61
|
+
|
|
62
|
+
Run evals inside the image in [container/](../container/Dockerfile), which
|
|
63
|
+
follows promptfoo's guidance for
|
|
64
|
+
[coding agents](https://www.promptfoo.dev/docs/guides/evaluate-coding-agents/):
|
|
65
|
+
a workspace is not a sandbox, so agents run in an ephemeral container with a
|
|
66
|
+
read-only mount of the repository, writes on a separate volume, a
|
|
67
|
+
project-local Codex home, and only the credentials they need. On the host the
|
|
68
|
+
agent shares the operator's machine: it can read installed skills, enabled
|
|
69
|
+
plugins, global guidance, and any file by absolute path. `run` and `sweep` warn
|
|
70
|
+
when they are not isolated.
|
|
71
|
+
|
|
72
|
+
```sh
|
|
73
|
+
docker build -t skillcheck:<version> --build-arg SKILLCHECK_VERSION=<version> container
|
|
74
|
+
docker run --rm --env-file <owner-only env file> \
|
|
75
|
+
-v "$PWD:/srv/work:ro" -v "$PWD/.skillcheck:/srv/work/.skillcheck" \
|
|
76
|
+
-v <codex config.toml>:/etc/skillcheck/codex/config.toml:ro \
|
|
77
|
+
skillcheck:<version> run --root /srv/work skills/<skill>/evals/<scenario> --trials 3
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
skillcheck runs as root and reads the repository from `/srv`, which only root
|
|
81
|
+
can enter. Each agent binary starts through a wrapper that drops to the
|
|
82
|
+
unprivileged `agent` user, which owns only its workdir and the run's homes; the
|
|
83
|
+
run's promptfoo config and manifests, which carry the criteria, and promptfoo's
|
|
84
|
+
own state stay root-only. Start one container per `run` or `sweep`
|
|
85
|
+
invocation, as above: agents in one container share the agent user, so a
|
|
86
|
+
control must never share a container with a skill run. The image's home is
|
|
87
|
+
empty. Pass Claude gateway auth
|
|
88
|
+
through the env file ([auth](#auth)) and mount a Codex `config.toml` written for
|
|
89
|
+
the run that names only the model provider and its auth, never the operator's
|
|
90
|
+
own; a token its auth command reads must be readable by the agent user. Codex's
|
|
91
|
+
namespace sandbox cannot start inside a container, so skillcheck runs it with
|
|
92
|
+
full access there; the container and the user drop are the boundary. A run
|
|
93
|
+
counts as isolated only inside a container, with the image's marker variable,
|
|
94
|
+
skillcheck running as root, an agent user configured, and the wrapper
|
|
95
|
+
installed; results record `isolated` in their run
|
|
96
|
+
configuration, so `summarize` refuses to mix isolated and host rows and `sweep`
|
|
97
|
+
reruns host results.
|
|
98
|
+
|
|
60
99
|
### Trials
|
|
61
100
|
|
|
62
101
|
One trial is one sample of a noisy process: the same scenario and skill can
|
|
@@ -221,6 +260,7 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
|
|
|
221
260
|
"judge_effort": null,
|
|
222
261
|
"trials": 3,
|
|
223
262
|
"agent_access": "online",
|
|
263
|
+
"isolated": true,
|
|
224
264
|
"aggregate": { "pass": false, "pass_rate": 0.6667, "score": 0.81, "...": "..." },
|
|
225
265
|
"ran_at": "<ISO timestamp>",
|
|
226
266
|
"tool_version": "<skillcheck version>"
|
|
@@ -228,8 +268,8 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
|
|
|
228
268
|
```
|
|
229
269
|
|
|
230
270
|
`summarize` reads those sidecars and refuses to mix skills-tree revisions or
|
|
231
|
-
run configurations (agent model and effort, judge model and effort, trials,
|
|
232
|
-
agent access) in
|
|
271
|
+
run configurations (agent model and effort, judge model and effort, trials,
|
|
272
|
+
agent access, and isolation) in
|
|
233
273
|
one scorecard, including retained rows from partial reruns, unless
|
|
234
274
|
`--allow-mixed`. Configurations are compared within a harness, since harnesses
|
|
235
275
|
differ by design. Sidecars written before run configurations were recorded fall
|