@uinaf/skillcheck 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -28,6 +28,7 @@ function parseArgs(argv) {
28
28
  "--root",
29
29
  "--agent",
30
30
  "--judge",
31
+ "--judge-effort",
31
32
  "--harness",
32
33
  "--max-turns"
33
34
  ]);
@@ -60,10 +61,23 @@ function stateDirs(root) {
60
61
  function runOptions(flags) {
61
62
  const harness = flags.get("--harness") ?? "claude";
62
63
  if (harness !== "claude" && harness !== "codex" && harness !== "cursor") fail(`--harness must be claude, codex, or cursor, got ${harness}`);
64
+ const agent = flags.get("--agent");
65
+ const judgeModel = flags.get("--judge") ?? "claude-opus-5";
66
+ const judgeEffort = flags.get("--judge-effort");
67
+ if (judgeEffort !== void 0) {
68
+ if (![
69
+ "minimal",
70
+ "low",
71
+ "medium",
72
+ "high"
73
+ ].includes(judgeEffort)) fail(`--judge-effort must be minimal, low, medium, or high, got ${judgeEffort}`);
74
+ if (!judgeModel.includes(":")) fail("--judge-effort needs a provider-qualified --judge (e.g. openai:chat:gpt-5.6-sol); the Anthropic judge does not take a reasoning effort");
75
+ }
63
76
  return {
64
77
  harness,
65
- agentModel: flags.get("--agent"),
66
- judgeModel: flags.get("--judge") ?? "claude-opus-5",
78
+ agentModel: agent,
79
+ judgeModel,
80
+ judgeEffort,
67
81
  maxTurns: flags.has("--max-turns") ? parseMaxTurns(flags.get("--max-turns")) : void 0
68
82
  };
69
83
  }
@@ -72,8 +86,9 @@ function classifyResult(raw) {
72
86
  const res = root?.results?.results?.[0];
73
87
  if (res === void 0) return { error: "promptfoo output carried no result" };
74
88
  const message = typeof res.error === "string" ? res.error.trim() : "";
75
- if (message !== "") return { error: message };
76
89
  const stats = root?.results?.stats;
90
+ const gradedByStats = stats !== void 0 && (stats.errors ?? 0) === 0 && ((stats.failures ?? 0) > 0 || (stats.successes ?? 0) > 0);
91
+ if (message !== "" && !gradedByStats) return { error: message };
77
92
  if (stats !== void 0 && (stats.errors ?? 0) > 0 && (stats.successes ?? 0) === 0 && (stats.failures ?? 0) === 0) return { error: "promptfoo reported an errored test with nothing graded" };
78
93
  if (typeof res.score !== "number" || typeof res.success !== "boolean") return { error: "promptfoo result carried no usable score" };
79
94
  return {
@@ -166,7 +181,7 @@ function discoverScenarios(root) {
166
181
  }
167
182
  function cmdRun(argv) {
168
183
  const { positional, flags } = parseArgs(argv);
169
- if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--harness claude|codex|cursor]");
184
+ if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--judge-effort EFFORT] [--harness claude|codex|cursor]");
170
185
  const o = runScenario(positional[0], runOptions(flags), resolveRoot(flags));
171
186
  if (o.score === void 0) {
172
187
  console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
@@ -242,7 +257,7 @@ function reduceResults(dir, allowMixed) {
242
257
  score: res.score,
243
258
  pass: res.success,
244
259
  agent_model: provider?.config?.model ?? `${harness}-default`,
245
- judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? "unknown",
260
+ judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? judge?.id ?? "unknown",
246
261
  latency_ms: res.latencyMs,
247
262
  tokens: (res.tokenUsage?.total ?? 0) + (res.tokenUsage?.assertions?.total ?? 0)
248
263
  });
package/dist/scenario.js CHANGED
@@ -155,7 +155,10 @@ function buildConfig(s, workdir, manifestPath, opts, paths) {
155
155
  prompts: ["{{task}}"],
156
156
  providers: [agentProvider(opts, workdir, s.skill, paths)],
157
157
  defaultTest: { options: {
158
- provider: process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
158
+ provider: opts.judgeModel.includes(":") ? opts.judgeEffort === void 0 ? opts.judgeModel : {
159
+ id: opts.judgeModel,
160
+ config: { reasoning_effort: opts.judgeEffort }
161
+ } : process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
159
162
  id: "anthropic:claude-agent-sdk",
160
163
  config: {
161
164
  model: opts.judgeModel,
package/docs/usage.md CHANGED
@@ -56,6 +56,20 @@ replays the CLI's `stream-json` output: the `result` event becomes the graded
56
56
  output and `SKILL.md` reads under `.cursor/skills/` become the `skill-used`
57
57
  evidence. the judge leg is unchanged.
58
58
 
59
+ `--judge` takes either a bare Claude model (graded through the Anthropic
60
+ selection in [auth](#auth)) or a provider-qualified promptfoo id, passed
61
+ through verbatim:
62
+
63
+ ```sh
64
+ skillcheck run <scenario-dir> --judge openai:chat:gpt-5.6-sol --judge-effort high
65
+ ```
66
+
67
+ a provider-qualified judge authenticates through that provider's own env
68
+ (`OPENAI_API_KEY`, plus `OPENAI_BASE_URL` for a gateway) and is recorded
69
+ verbatim in the scorecard's `judge_model` column. `--judge-effort`
70
+ (minimal|low|medium|high) sets `reasoning_effort` and requires a
71
+ provider-qualified judge; the Anthropic judge does not take one.
72
+
59
73
  ## sweep
60
74
 
61
75
  ```sh
@@ -130,5 +144,7 @@ written inside the installed package.
130
144
  | `CODEX_HOME` (default `~/.codex`) | where the codex harness finds the local `codex` CLI login |
131
145
  | `OPENAI_API_KEY` | codex agent auth when there is no local login |
132
146
  | `CURSOR_API_KEY` | cursor agent auth; a logged-in `cursor-agent` also works |
147
+ | `OPENAI_API_KEY` + `OPENAI_BASE_URL` | a provider-qualified `--judge openai:…`, optionally via a gateway |
133
148
 
134
- the judge stays on the Anthropic selection regardless of the agent harness.
149
+ a bare `--judge` model stays on the Anthropic selection regardless of the
150
+ agent harness; a provider-qualified `--judge` uses that provider's env instead.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "0.3.0",
3
+ "version": "0.4.0",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {