@uinaf/skillcheck 1.1.0 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +7 -5
- package/dist/scenario.js +5 -1
- package/docs/usage.md +10 -3
- package/package.json +1 -1
package/dist/cli.js
CHANGED
|
@@ -80,13 +80,13 @@ function runOptions(flags) {
|
|
|
80
80
|
const judgeModel = flags.get("--judge") ?? "claude-opus-5";
|
|
81
81
|
const judgeEffort = flags.get("--judge-effort");
|
|
82
82
|
if (judgeEffort !== void 0) {
|
|
83
|
-
|
|
83
|
+
const levels = judgeModel.includes(":") ? [
|
|
84
84
|
"minimal",
|
|
85
85
|
"low",
|
|
86
86
|
"medium",
|
|
87
87
|
"high"
|
|
88
|
-
]
|
|
89
|
-
if (!
|
|
88
|
+
] : AGENT_EFFORTS;
|
|
89
|
+
if (!levels.includes(judgeEffort)) throw new Error(`--judge-effort for ${judgeModel} must be ${levels.join(", ")}, got ${judgeEffort}`);
|
|
90
90
|
}
|
|
91
91
|
return {
|
|
92
92
|
harness,
|
|
@@ -304,7 +304,8 @@ function judgeName(judge) {
|
|
|
304
304
|
if (typeof judge === "string") return judge.replace(/^anthropic:messages:/, "");
|
|
305
305
|
const j = judge;
|
|
306
306
|
const name = j?.config?.model ?? j?.id;
|
|
307
|
-
|
|
307
|
+
if (typeof name !== "string") return "unknown";
|
|
308
|
+
return j?.config?.effort === void 0 ? name : name.replace(/^anthropic:messages:/, "");
|
|
308
309
|
}
|
|
309
310
|
function resultRunConfig(raw, meta, harness) {
|
|
310
311
|
const m = meta;
|
|
@@ -318,7 +319,8 @@ function resultRunConfig(raw, meta, harness) {
|
|
|
318
319
|
const r = raw;
|
|
319
320
|
const agent = r?.config?.providers?.[0]?.config;
|
|
320
321
|
const judge = r?.config?.defaultTest?.options?.provider;
|
|
321
|
-
const
|
|
322
|
+
const judgeConfig = judge?.config;
|
|
323
|
+
const judgeEffort = judgeConfig?.reasoning_effort ?? judgeConfig?.effort;
|
|
322
324
|
return {
|
|
323
325
|
agent_model: typeof agent?.model === "string" ? agent.model : `${harness}-default`,
|
|
324
326
|
agent_effort: typeof agent?.effort === "string" ? agent.effort : null,
|
package/dist/scenario.js
CHANGED
|
@@ -179,10 +179,14 @@ function buildConfig(s, trials, opts, paths) {
|
|
|
179
179
|
provider: opts.judgeModel.includes(":") ? opts.judgeEffort === void 0 ? opts.judgeModel : {
|
|
180
180
|
id: opts.judgeModel,
|
|
181
181
|
config: { reasoning_effort: opts.judgeEffort }
|
|
182
|
-
} : process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
|
|
182
|
+
} : process.env.ANTHROPIC_API_KEY ? opts.judgeEffort === void 0 ? `anthropic:messages:${opts.judgeModel}` : {
|
|
183
|
+
id: `anthropic:messages:${opts.judgeModel}`,
|
|
184
|
+
config: { effort: opts.judgeEffort }
|
|
185
|
+
} : {
|
|
183
186
|
id: "anthropic:claude-agent-sdk",
|
|
184
187
|
config: {
|
|
185
188
|
model: opts.judgeModel,
|
|
189
|
+
...opts.judgeEffort ? { effort: opts.judgeEffort } : {},
|
|
186
190
|
apiKeyRequired: false,
|
|
187
191
|
max_turns: 3,
|
|
188
192
|
output_format: {
|
package/docs/usage.md
CHANGED
|
@@ -109,9 +109,16 @@ skillcheck run <scenario-dir> --judge openai:chat:gpt-5.6-sol --judge-effort hig
|
|
|
109
109
|
|
|
110
110
|
A provider-qualified judge authenticates through that provider's own env
|
|
111
111
|
(`OPENAI_API_KEY`, plus `OPENAI_BASE_URL` for a gateway) and is recorded
|
|
112
|
-
verbatim in the scorecard's `judge_model` column. `--judge-effort`
|
|
113
|
-
(minimal|low|medium|high) sets `reasoning_effort
|
|
114
|
-
|
|
112
|
+
verbatim in the scorecard's `judge_model` column. For it, `--judge-effort`
|
|
113
|
+
(minimal|low|medium|high) sets `reasoning_effort`. For a bare Claude judge,
|
|
114
|
+
`--judge-effort` takes Claude's levels (low|medium|high|xhigh|max) and is passed
|
|
115
|
+
as `effort` on either Anthropic path; the SDK judge starts Claude Code with
|
|
116
|
+
`--effort`:
|
|
117
|
+
|
|
118
|
+
```sh
|
|
119
|
+
skillcheck run <scenario-dir> --agent claude-opus-5-5 --agent-effort medium \
|
|
120
|
+
--judge claude-opus-5-5 --judge-effort high --trials 3
|
|
121
|
+
```
|
|
115
122
|
|
|
116
123
|
## Sweep
|
|
117
124
|
|