@uinaf/skillcheck 0.3.1 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +18 -4
- package/dist/scenario.js +4 -1
- package/docs/usage.md +17 -1
- package/package.json +1 -1
package/dist/cli.js
CHANGED
|
@@ -28,6 +28,7 @@ function parseArgs(argv) {
|
|
|
28
28
|
"--root",
|
|
29
29
|
"--agent",
|
|
30
30
|
"--judge",
|
|
31
|
+
"--judge-effort",
|
|
31
32
|
"--harness",
|
|
32
33
|
"--max-turns"
|
|
33
34
|
]);
|
|
@@ -60,10 +61,23 @@ function stateDirs(root) {
|
|
|
60
61
|
function runOptions(flags) {
|
|
61
62
|
const harness = flags.get("--harness") ?? "claude";
|
|
62
63
|
if (harness !== "claude" && harness !== "codex" && harness !== "cursor") fail(`--harness must be claude, codex, or cursor, got ${harness}`);
|
|
64
|
+
const agent = flags.get("--agent");
|
|
65
|
+
const judgeModel = flags.get("--judge") ?? "claude-opus-5";
|
|
66
|
+
const judgeEffort = flags.get("--judge-effort");
|
|
67
|
+
if (judgeEffort !== void 0) {
|
|
68
|
+
if (![
|
|
69
|
+
"minimal",
|
|
70
|
+
"low",
|
|
71
|
+
"medium",
|
|
72
|
+
"high"
|
|
73
|
+
].includes(judgeEffort)) fail(`--judge-effort must be minimal, low, medium, or high, got ${judgeEffort}`);
|
|
74
|
+
if (!judgeModel.includes(":")) fail("--judge-effort needs a provider-qualified --judge (e.g. openai:chat:gpt-5.6-sol); the Anthropic judge does not take a reasoning effort");
|
|
75
|
+
}
|
|
63
76
|
return {
|
|
64
77
|
harness,
|
|
65
|
-
agentModel:
|
|
66
|
-
judgeModel
|
|
78
|
+
agentModel: agent,
|
|
79
|
+
judgeModel,
|
|
80
|
+
judgeEffort,
|
|
67
81
|
maxTurns: flags.has("--max-turns") ? parseMaxTurns(flags.get("--max-turns")) : void 0
|
|
68
82
|
};
|
|
69
83
|
}
|
|
@@ -167,7 +181,7 @@ function discoverScenarios(root) {
|
|
|
167
181
|
}
|
|
168
182
|
function cmdRun(argv) {
|
|
169
183
|
const { positional, flags } = parseArgs(argv);
|
|
170
|
-
if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--harness claude|codex|cursor]");
|
|
184
|
+
if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--judge-effort EFFORT] [--harness claude|codex|cursor]");
|
|
171
185
|
const o = runScenario(positional[0], runOptions(flags), resolveRoot(flags));
|
|
172
186
|
if (o.score === void 0) {
|
|
173
187
|
console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
|
|
@@ -243,7 +257,7 @@ function reduceResults(dir, allowMixed) {
|
|
|
243
257
|
score: res.score,
|
|
244
258
|
pass: res.success,
|
|
245
259
|
agent_model: provider?.config?.model ?? `${harness}-default`,
|
|
246
|
-
judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? "unknown",
|
|
260
|
+
judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? judge?.id ?? "unknown",
|
|
247
261
|
latency_ms: res.latencyMs,
|
|
248
262
|
tokens: (res.tokenUsage?.total ?? 0) + (res.tokenUsage?.assertions?.total ?? 0)
|
|
249
263
|
});
|
package/dist/scenario.js
CHANGED
|
@@ -155,7 +155,10 @@ function buildConfig(s, workdir, manifestPath, opts, paths) {
|
|
|
155
155
|
prompts: ["{{task}}"],
|
|
156
156
|
providers: [agentProvider(opts, workdir, s.skill, paths)],
|
|
157
157
|
defaultTest: { options: {
|
|
158
|
-
provider:
|
|
158
|
+
provider: opts.judgeModel.includes(":") ? opts.judgeEffort === void 0 ? opts.judgeModel : {
|
|
159
|
+
id: opts.judgeModel,
|
|
160
|
+
config: { reasoning_effort: opts.judgeEffort }
|
|
161
|
+
} : process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
|
|
159
162
|
id: "anthropic:claude-agent-sdk",
|
|
160
163
|
config: {
|
|
161
164
|
model: opts.judgeModel,
|
package/docs/usage.md
CHANGED
|
@@ -56,6 +56,20 @@ replays the CLI's `stream-json` output: the `result` event becomes the graded
|
|
|
56
56
|
output and `SKILL.md` reads under `.cursor/skills/` become the `skill-used`
|
|
57
57
|
evidence. the judge leg is unchanged.
|
|
58
58
|
|
|
59
|
+
`--judge` takes either a bare Claude model (graded through the Anthropic
|
|
60
|
+
selection in [auth](#auth)) or a provider-qualified promptfoo id, passed
|
|
61
|
+
through verbatim:
|
|
62
|
+
|
|
63
|
+
```sh
|
|
64
|
+
skillcheck run <scenario-dir> --judge openai:chat:gpt-5.6-sol --judge-effort high
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
a provider-qualified judge authenticates through that provider's own env
|
|
68
|
+
(`OPENAI_API_KEY`, plus `OPENAI_BASE_URL` for a gateway) and is recorded
|
|
69
|
+
verbatim in the scorecard's `judge_model` column. `--judge-effort`
|
|
70
|
+
(minimal|low|medium|high) sets `reasoning_effort` and requires a
|
|
71
|
+
provider-qualified judge; the Anthropic judge does not take one.
|
|
72
|
+
|
|
59
73
|
## sweep
|
|
60
74
|
|
|
61
75
|
```sh
|
|
@@ -130,5 +144,7 @@ written inside the installed package.
|
|
|
130
144
|
| `CODEX_HOME` (default `~/.codex`) | where the codex harness finds the local `codex` CLI login |
|
|
131
145
|
| `OPENAI_API_KEY` | codex agent auth when there is no local login |
|
|
132
146
|
| `CURSOR_API_KEY` | cursor agent auth; a logged-in `cursor-agent` also works |
|
|
147
|
+
| `OPENAI_API_KEY` + `OPENAI_BASE_URL` | a provider-qualified `--judge openai:…`, optionally via a gateway |
|
|
133
148
|
|
|
134
|
-
|
|
149
|
+
a bare `--judge` model stays on the Anthropic selection regardless of the
|
|
150
|
+
agent harness; a provider-qualified `--judge` uses that provider's env instead.
|