@uinaf/skillcheck 1.5.3 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -1,5 +1,5 @@
1
1
  #!/usr/bin/env node
2
- import { encodeRunNamePart, generateRun, requiredEvalPackages, resolvePackageDir, runNameFor } from "./scenario.js";
2
+ import { encodeRunNamePart, generateRun, isIsolated, requiredEvalPackages, resolvePackageDir, runNameFor } from "./scenario.js";
3
3
  import { lintSkills } from "./lint.js";
4
4
  import { execFileSync, spawnSync } from "node:child_process";
5
5
  import { createHash } from "node:crypto";
@@ -110,7 +110,8 @@ function runConfigOf(opts) {
110
110
  judge_model: opts.judgeModel,
111
111
  judge_effort: opts.judgeEffort ?? null,
112
112
  trials: opts.trials ?? 1,
113
- agent_access: "online"
113
+ agent_access: "online",
114
+ isolated: isIsolated()
114
115
  };
115
116
  }
116
117
  function configKey(c) {
@@ -120,12 +121,13 @@ function configKey(c) {
120
121
  c.judge_model,
121
122
  c.judge_effort ?? null,
122
123
  c.trials ?? 1,
123
- c.agent_access ?? "offline"
124
+ c.agent_access ?? "offline",
125
+ c.isolated ?? false
124
126
  ]);
125
127
  }
126
128
  function describeConfig(c) {
127
129
  const effort = (e) => e ? `@${e}` : "";
128
- return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}, ${c.agent_access ?? "offline"}`;
130
+ return `agent ${c.agent_model}${effort(c.agent_effort)}, judge ${c.judge_model}${effort(c.judge_effort)}, trials ${c.trials ?? 1}, ${c.agent_access ?? "offline"}, ${c.isolated ? "isolated" : "host"}`;
129
131
  }
130
132
  function assertUniformConfig(entries, allowMixed) {
131
133
  if (allowMixed) return;
@@ -335,7 +337,8 @@ function resultRunConfig(raw, meta, harness) {
335
337
  judge_model: m.judge_model,
336
338
  judge_effort: m.judge_effort ?? null,
337
339
  trials: m.trials ?? 1,
338
- agent_access: m.agent_access === "online" ? "online" : "offline"
340
+ agent_access: m.agent_access === "online" ? "online" : "offline",
341
+ isolated: m.isolated === true
339
342
  };
340
343
  const r = raw;
341
344
  const agent = r?.config?.providers?.[0]?.config;
@@ -348,7 +351,8 @@ function resultRunConfig(raw, meta, harness) {
348
351
  judge_model: judgeName(judge),
349
352
  judge_effort: typeof judgeEffort === "string" ? judgeEffort : null,
350
353
  trials: r?.results?.results?.length ?? 1,
351
- agent_access: "offline"
354
+ agent_access: "offline",
355
+ isolated: false
352
356
  };
353
357
  }
354
358
  function readJson(file) {
@@ -390,11 +394,16 @@ function discoverScenarios(root) {
390
394
  return found.sort();
391
395
  }
392
396
  const RUN_FLAGS = "[--root DIR] [--harness claude|codex|grok] [--agent MODEL] [--agent-effort EFFORT] [--judge MODEL] [--judge-effort EFFORT] [--trials K] [--max-turns N] [--control]";
397
+ function warnHostRun() {
398
+ if (isIsolated()) return;
399
+ console.error("warning: host run; the agent can read this machine's installed skills, plugins, and files, so scores are not isolated. Run in the isolated image: docs/usage.md#isolated-runs");
400
+ }
393
401
  function cmdRun(argv) {
394
402
  const { positional, flags } = parseArgs(argv);
395
403
  if (positional.length !== 1) fail(`usage: skillcheck run <scenario-dir> ${RUN_FLAGS}`);
396
404
  const opts = runOptions(flags);
397
405
  ensureEvalPackages(opts);
406
+ warnHostRun();
398
407
  const o = runScenario(positional[0], opts, resolveRoot(flags));
399
408
  if (o.stats === void 0) {
400
409
  console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
@@ -409,6 +418,7 @@ function cmdSweep(argv) {
409
418
  const root = resolveRoot(flags);
410
419
  const opts = runOptions(flags);
411
420
  ensureEvalPackages(opts);
421
+ warnHostRun();
412
422
  const all = flags.get("--all") === true;
413
423
  const resultsDir = stateDirs(root).results;
414
424
  const wanted = configKey(runConfigOf(opts));
package/dist/scenario.js CHANGED
@@ -75,6 +75,10 @@ function stripHiddenFlag(skillMd) {
75
75
  if (!range) return skillMd;
76
76
  return skillMd.split("\n").filter((l, i) => !(i >= range[0] && i < range[1] && l.startsWith("disable-model-invocation:"))).join("\n");
77
77
  }
78
+ function isIsolated() {
79
+ return process.env.SKILLCHECK_ISOLATED === "1" && (fs.existsSync("/.dockerenv") || fs.existsSync("/run/.containerenv")) && process.getuid?.() === 0 && Number.isInteger(Number(process.env.SKILLCHECK_AGENT_UID)) && fs.existsSync(AGENT_WRAPPER);
80
+ }
81
+ const AGENT_WRAPPER = "/usr/local/libexec/skillcheck/agent-wrapper.sh";
78
82
  const MARKDOWN_LINK = /\]\(\s*<([^>]+)>|\]\(\s*([^)\s]+)|^ {0,3}\[[^\]]+\]:\s*<?([^\s>]+)/gm;
79
83
  function linkedSiblings(skillDir) {
80
84
  const skillsRoot = path.dirname(skillDir);
@@ -167,7 +171,7 @@ function materialize(s, runDir, harness, control = false) {
167
171
  };
168
172
  walk(workdir);
169
173
  const manifestPath = path.join(runDir, "manifest.json");
170
- fs.writeFileSync(manifestPath, JSON.stringify(manifest, null, 2));
174
+ fs.writeFileSync(manifestPath, JSON.stringify(manifest, null, 2), { mode: 384 });
171
175
  return {
172
176
  workdir,
173
177
  manifestPath
@@ -189,7 +193,7 @@ function agentProvider(opts, workdir, skill, paths) {
189
193
  working_dir: workdir,
190
194
  skip_git_repo_check: true,
191
195
  enable_streaming: true,
192
- sandbox_mode: "workspace-write",
196
+ sandbox_mode: isIsolated() ? "danger-full-access" : "workspace-write",
193
197
  network_access_enabled: true,
194
198
  web_search_enabled: true,
195
199
  cli_config: { features: { plugins: false } },
@@ -358,6 +362,24 @@ function privateHome(dir) {
358
362
  fs.symlinkSync(path.join(source, name), path.join(dir, name));
359
363
  }
360
364
  }
365
+ function handOverToAgent(runDir, trials) {
366
+ const uid = Number(process.env.SKILLCHECK_AGENT_UID);
367
+ const gid = Number(process.env.SKILLCHECK_AGENT_GID ?? process.env.SKILLCHECK_AGENT_UID);
368
+ if (!Number.isInteger(uid) || !Number.isInteger(gid)) throw new Error("isolated run without SKILLCHECK_AGENT_UID: refusing to run the agent as root");
369
+ const own = (p) => {
370
+ fs.lchownSync(p, uid, gid);
371
+ if (fs.lstatSync(p).isDirectory()) for (const e of fs.readdirSync(p)) own(path.join(p, e));
372
+ };
373
+ for (const dir of [path.dirname(runDir), runDir]) fs.chmodSync(dir, 457);
374
+ for (const t of trials) {
375
+ fs.chmodSync(path.dirname(t.workdir), 457);
376
+ own(t.workdir);
377
+ }
378
+ for (const home of ["codex-home", "home"]) {
379
+ const p = path.join(runDir, home);
380
+ if (fs.existsSync(p)) own(p);
381
+ }
382
+ }
361
383
  function generateRun(scenarioDir, opts, paths) {
362
384
  const s = loadScenario(scenarioDir);
363
385
  const name = runNameFor(scenarioDir, opts.harness, opts.control);
@@ -382,7 +404,8 @@ function generateRun(scenarioDir, opts, paths) {
382
404
  }
383
405
  const config = buildConfig(s, trials, opts, paths);
384
406
  const configPath = path.join(runDir, "promptfooconfig.json");
385
- fs.writeFileSync(configPath, JSON.stringify(config, null, 2));
407
+ fs.writeFileSync(configPath, JSON.stringify(config, null, 2), { mode: 384 });
408
+ if (isIsolated()) handOverToAgent(runDir, trials);
386
409
  return {
387
410
  name,
388
411
  configPath,
@@ -391,4 +414,4 @@ function generateRun(scenarioDir, opts, paths) {
391
414
  };
392
415
  }
393
416
  //#endregion
394
- export { DEFAULT_CLAUDE_AGENT, FILE_BLOCK, buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, linkedSiblings, loadScenario, materialize, privateCodexHome, privateHome, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
417
+ export { DEFAULT_CLAUDE_AGENT, FILE_BLOCK, buildConfig, encodeRunNamePart, generateRun, handOverToAgent, isHiddenSkill, isIsolated, linkedSiblings, loadScenario, materialize, privateCodexHome, privateHome, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag, trialLabel };
package/docs/usage.md CHANGED
@@ -57,6 +57,45 @@ and a Claude agent limit of 50 turns. `--max-turns` changes that limit only for
57
57
  Claude; passing it with `codex` or `grok` fails before the eval starts. On
58
58
  those harnesses, omitting `--agent` leaves the model to that CLI's own default.
59
59
 
60
+ ### Isolated runs
61
+
62
+ Run evals inside the image in [container/](../container/Dockerfile), which
63
+ follows promptfoo's guidance for
64
+ [coding agents](https://www.promptfoo.dev/docs/guides/evaluate-coding-agents/):
65
+ a workspace is not a sandbox, so agents run in an ephemeral container with a
66
+ read-only mount of the repository, writes on a separate volume, a
67
+ project-local Codex home, and only the credentials they need. On the host the
68
+ agent shares the operator's machine: it can read installed skills, enabled
69
+ plugins, global guidance, and any file by absolute path. `run` and `sweep` warn
70
+ when they are not isolated.
71
+
72
+ ```sh
73
+ docker build -t skillcheck:<version> --build-arg SKILLCHECK_VERSION=<version> container
74
+ docker run --rm --env-file <owner-only env file> \
75
+ -v "$PWD:/srv/work:ro" -v "$PWD/.skillcheck:/srv/work/.skillcheck" \
76
+ -v <codex config.toml>:/etc/skillcheck/codex/config.toml:ro \
77
+ skillcheck:<version> run --root /srv/work skills/<skill>/evals/<scenario> --trials 3
78
+ ```
79
+
80
+ skillcheck runs as root and reads the repository from `/srv`, which only root
81
+ can enter. Each agent binary starts through a wrapper that drops to the
82
+ unprivileged `agent` user, which owns only its workdir and the run's homes; the
83
+ run's promptfoo config and manifests, which carry the criteria, and promptfoo's
84
+ own state stay root-only. Start one container per `run` or `sweep`
85
+ invocation, as above: agents in one container share the agent user, so a
86
+ control must never share a container with a skill run. The image's home is
87
+ empty. Pass Claude gateway auth
88
+ through the env file ([auth](#auth)) and mount a Codex `config.toml` written for
89
+ the run that names only the model provider and its auth, never the operator's
90
+ own; a token its auth command reads must be readable by the agent user. Codex's
91
+ namespace sandbox cannot start inside a container, so skillcheck runs it with
92
+ full access there; the container and the user drop are the boundary. A run
93
+ counts as isolated only inside a container, with the image's marker variable,
94
+ skillcheck running as root, an agent user configured, and the wrapper
95
+ installed; results record `isolated` in their run
96
+ configuration, so `summarize` refuses to mix isolated and host rows and `sweep`
97
+ reruns host results.
98
+
60
99
  ### Trials
61
100
 
62
101
  One trial is one sample of a noisy process: the same scenario and skill can
@@ -221,6 +260,7 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
221
260
  "judge_effort": null,
222
261
  "trials": 3,
223
262
  "agent_access": "online",
263
+ "isolated": true,
224
264
  "aggregate": { "pass": false, "pass_rate": 0.6667, "score": 0.81, "...": "..." },
225
265
  "ran_at": "<ISO timestamp>",
226
266
  "tool_version": "<skillcheck version>"
@@ -228,8 +268,8 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
228
268
  ```
229
269
 
230
270
  `summarize` reads those sidecars and refuses to mix skills-tree revisions or
231
- run configurations (agent model and effort, judge model and effort, trials, and
232
- agent access) in
271
+ run configurations (agent model and effort, judge model and effort, trials,
272
+ agent access, and isolation) in
233
273
  one scorecard, including retained rows from partial reruns, unless
234
274
  `--allow-mixed`. Configurations are compared within a harness, since harnesses
235
275
  differ by design. Sidecars written before run configurations were recorded fall
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "1.5.3",
3
+ "version": "1.6.0",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {