@uinaf/skillcheck 0.3.1 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -28,6 +28,7 @@ function parseArgs(argv) {
28
28
  "--root",
29
29
  "--agent",
30
30
  "--judge",
31
+ "--judge-effort",
31
32
  "--harness",
32
33
  "--max-turns"
33
34
  ]);
@@ -60,10 +61,23 @@ function stateDirs(root) {
60
61
  function runOptions(flags) {
61
62
  const harness = flags.get("--harness") ?? "claude";
62
63
  if (harness !== "claude" && harness !== "codex" && harness !== "cursor") fail(`--harness must be claude, codex, or cursor, got ${harness}`);
64
+ const agent = flags.get("--agent");
65
+ const judgeModel = flags.get("--judge") ?? "claude-opus-5";
66
+ const judgeEffort = flags.get("--judge-effort");
67
+ if (judgeEffort !== void 0) {
68
+ if (![
69
+ "minimal",
70
+ "low",
71
+ "medium",
72
+ "high"
73
+ ].includes(judgeEffort)) fail(`--judge-effort must be minimal, low, medium, or high, got ${judgeEffort}`);
74
+ if (!judgeModel.includes(":")) fail("--judge-effort needs a provider-qualified --judge (e.g. openai:chat:gpt-5.6-sol); the Anthropic judge does not take a reasoning effort");
75
+ }
63
76
  return {
64
77
  harness,
65
- agentModel: flags.get("--agent"),
66
- judgeModel: flags.get("--judge") ?? "claude-opus-5",
78
+ agentModel: agent,
79
+ judgeModel,
80
+ judgeEffort,
67
81
  maxTurns: flags.has("--max-turns") ? parseMaxTurns(flags.get("--max-turns")) : void 0
68
82
  };
69
83
  }
@@ -167,7 +181,7 @@ function discoverScenarios(root) {
167
181
  }
168
182
  function cmdRun(argv) {
169
183
  const { positional, flags } = parseArgs(argv);
170
- if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--harness claude|codex|cursor]");
184
+ if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--judge-effort EFFORT] [--harness claude|codex|cursor]");
171
185
  const o = runScenario(positional[0], runOptions(flags), resolveRoot(flags));
172
186
  if (o.score === void 0) {
173
187
  console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
@@ -243,7 +257,7 @@ function reduceResults(dir, allowMixed) {
243
257
  score: res.score,
244
258
  pass: res.success,
245
259
  agent_model: provider?.config?.model ?? `${harness}-default`,
246
- judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? "unknown",
260
+ judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? judge?.id ?? "unknown",
247
261
  latency_ms: res.latencyMs,
248
262
  tokens: (res.tokenUsage?.total ?? 0) + (res.tokenUsage?.assertions?.total ?? 0)
249
263
  });
package/dist/scenario.js CHANGED
@@ -155,7 +155,10 @@ function buildConfig(s, workdir, manifestPath, opts, paths) {
155
155
  prompts: ["{{task}}"],
156
156
  providers: [agentProvider(opts, workdir, s.skill, paths)],
157
157
  defaultTest: { options: {
158
- provider: process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
158
+ provider: opts.judgeModel.includes(":") ? opts.judgeEffort === void 0 ? opts.judgeModel : {
159
+ id: opts.judgeModel,
160
+ config: { reasoning_effort: opts.judgeEffort }
161
+ } : process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
159
162
  id: "anthropic:claude-agent-sdk",
160
163
  config: {
161
164
  model: opts.judgeModel,
package/docs/usage.md CHANGED
@@ -56,6 +56,20 @@ replays the CLI's `stream-json` output: the `result` event becomes the graded
56
56
  output and `SKILL.md` reads under `.cursor/skills/` become the `skill-used`
57
57
  evidence. the judge leg is unchanged.
58
58
 
59
+ `--judge` takes either a bare Claude model (graded through the Anthropic
60
+ selection in [auth](#auth)) or a provider-qualified promptfoo id, passed
61
+ through verbatim:
62
+
63
+ ```sh
64
+ skillcheck run <scenario-dir> --judge openai:chat:gpt-5.6-sol --judge-effort high
65
+ ```
66
+
67
+ a provider-qualified judge authenticates through that provider's own env
68
+ (`OPENAI_API_KEY`, plus `OPENAI_BASE_URL` for a gateway) and is recorded
69
+ verbatim in the scorecard's `judge_model` column. `--judge-effort`
70
+ (minimal|low|medium|high) sets `reasoning_effort` and requires a
71
+ provider-qualified judge; the Anthropic judge does not take one.
72
+
59
73
  ## sweep
60
74
 
61
75
  ```sh
@@ -130,5 +144,7 @@ written inside the installed package.
130
144
  | `CODEX_HOME` (default `~/.codex`) | where the codex harness finds the local `codex` CLI login |
131
145
  | `OPENAI_API_KEY` | codex agent auth when there is no local login |
132
146
  | `CURSOR_API_KEY` | cursor agent auth; a logged-in `cursor-agent` also works |
147
+ | `OPENAI_API_KEY` + `OPENAI_BASE_URL` | a provider-qualified `--judge openai:…`, optionally via a gateway |
133
148
 
134
- the judge stays on the Anthropic selection regardless of the agent harness.
149
+ a bare `--judge` model stays on the Anthropic selection regardless of the
150
+ agent harness; a provider-qualified `--judge` uses that provider's env instead.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "0.3.1",
3
+ "version": "0.4.0",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {