@uinaf/skillcheck 1.1.0 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -80,13 +80,13 @@ function runOptions(flags) {
80
80
  const judgeModel = flags.get("--judge") ?? "claude-opus-5";
81
81
  const judgeEffort = flags.get("--judge-effort");
82
82
  if (judgeEffort !== void 0) {
83
- if (![
83
+ const levels = judgeModel.includes(":") ? [
84
84
  "minimal",
85
85
  "low",
86
86
  "medium",
87
87
  "high"
88
- ].includes(judgeEffort)) throw new Error(`--judge-effort must be minimal, low, medium, or high, got ${judgeEffort}`);
89
- if (!judgeModel.includes(":")) throw new Error("--judge-effort needs a provider-qualified --judge (e.g. openai:chat:gpt-5.6-sol); the Anthropic judge does not take a reasoning effort");
88
+ ] : AGENT_EFFORTS;
89
+ if (!levels.includes(judgeEffort)) throw new Error(`--judge-effort for ${judgeModel} must be ${levels.join(", ")}, got ${judgeEffort}`);
90
90
  }
91
91
  return {
92
92
  harness,
@@ -304,7 +304,8 @@ function judgeName(judge) {
304
304
  if (typeof judge === "string") return judge.replace(/^anthropic:messages:/, "");
305
305
  const j = judge;
306
306
  const name = j?.config?.model ?? j?.id;
307
- return typeof name === "string" ? name : "unknown";
307
+ if (typeof name !== "string") return "unknown";
308
+ return j?.config?.effort === void 0 ? name : name.replace(/^anthropic:messages:/, "");
308
309
  }
309
310
  function resultRunConfig(raw, meta, harness) {
310
311
  const m = meta;
@@ -318,7 +319,8 @@ function resultRunConfig(raw, meta, harness) {
318
319
  const r = raw;
319
320
  const agent = r?.config?.providers?.[0]?.config;
320
321
  const judge = r?.config?.defaultTest?.options?.provider;
321
- const judgeEffort = judge?.config?.reasoning_effort;
322
+ const judgeConfig = judge?.config;
323
+ const judgeEffort = judgeConfig?.reasoning_effort ?? judgeConfig?.effort;
322
324
  return {
323
325
  agent_model: typeof agent?.model === "string" ? agent.model : `${harness}-default`,
324
326
  agent_effort: typeof agent?.effort === "string" ? agent.effort : null,
package/dist/scenario.js CHANGED
@@ -179,10 +179,14 @@ function buildConfig(s, trials, opts, paths) {
179
179
  provider: opts.judgeModel.includes(":") ? opts.judgeEffort === void 0 ? opts.judgeModel : {
180
180
  id: opts.judgeModel,
181
181
  config: { reasoning_effort: opts.judgeEffort }
182
- } : process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
182
+ } : process.env.ANTHROPIC_API_KEY ? opts.judgeEffort === void 0 ? `anthropic:messages:${opts.judgeModel}` : {
183
+ id: `anthropic:messages:${opts.judgeModel}`,
184
+ config: { effort: opts.judgeEffort }
185
+ } : {
183
186
  id: "anthropic:claude-agent-sdk",
184
187
  config: {
185
188
  model: opts.judgeModel,
189
+ ...opts.judgeEffort ? { effort: opts.judgeEffort } : {},
186
190
  apiKeyRequired: false,
187
191
  max_turns: 3,
188
192
  output_format: {
package/docs/usage.md CHANGED
@@ -109,9 +109,16 @@ skillcheck run <scenario-dir> --judge openai:chat:gpt-5.6-sol --judge-effort hig
109
109
 
110
110
  A provider-qualified judge authenticates through that provider's own env
111
111
  (`OPENAI_API_KEY`, plus `OPENAI_BASE_URL` for a gateway) and is recorded
112
- verbatim in the scorecard's `judge_model` column. `--judge-effort`
113
- (minimal|low|medium|high) sets `reasoning_effort` and requires a
114
- provider-qualified judge; the Anthropic judge does not take one.
112
+ verbatim in the scorecard's `judge_model` column. For it, `--judge-effort`
113
+ (minimal|low|medium|high) sets `reasoning_effort`. For a bare Claude judge,
114
+ `--judge-effort` takes Claude's levels (low|medium|high|xhigh|max) and is passed
115
+ as `effort` on either Anthropic path; the SDK judge starts Claude Code with
116
+ `--effort`:
117
+
118
+ ```sh
119
+ skillcheck run <scenario-dir> --agent claude-opus-5-5 --agent-effort medium \
120
+ --judge claude-opus-5-5 --judge-effort high --trials 3
121
+ ```
115
122
 
116
123
  ## Sweep
117
124
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "1.1.0",
3
+ "version": "1.2.0",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {