@uinaf/skillcheck 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +20 -5
- package/dist/scenario.js +4 -1
- package/docs/usage.md +17 -1
- package/package.json +1 -1
package/dist/cli.js
CHANGED
|
@@ -28,6 +28,7 @@ function parseArgs(argv) {
|
|
|
28
28
|
"--root",
|
|
29
29
|
"--agent",
|
|
30
30
|
"--judge",
|
|
31
|
+
"--judge-effort",
|
|
31
32
|
"--harness",
|
|
32
33
|
"--max-turns"
|
|
33
34
|
]);
|
|
@@ -60,10 +61,23 @@ function stateDirs(root) {
|
|
|
60
61
|
function runOptions(flags) {
|
|
61
62
|
const harness = flags.get("--harness") ?? "claude";
|
|
62
63
|
if (harness !== "claude" && harness !== "codex" && harness !== "cursor") fail(`--harness must be claude, codex, or cursor, got ${harness}`);
|
|
64
|
+
const agent = flags.get("--agent");
|
|
65
|
+
const judgeModel = flags.get("--judge") ?? "claude-opus-5";
|
|
66
|
+
const judgeEffort = flags.get("--judge-effort");
|
|
67
|
+
if (judgeEffort !== void 0) {
|
|
68
|
+
if (![
|
|
69
|
+
"minimal",
|
|
70
|
+
"low",
|
|
71
|
+
"medium",
|
|
72
|
+
"high"
|
|
73
|
+
].includes(judgeEffort)) fail(`--judge-effort must be minimal, low, medium, or high, got ${judgeEffort}`);
|
|
74
|
+
if (!judgeModel.includes(":")) fail("--judge-effort needs a provider-qualified --judge (e.g. openai:chat:gpt-5.6-sol); the Anthropic judge does not take a reasoning effort");
|
|
75
|
+
}
|
|
63
76
|
return {
|
|
64
77
|
harness,
|
|
65
|
-
agentModel:
|
|
66
|
-
judgeModel
|
|
78
|
+
agentModel: agent,
|
|
79
|
+
judgeModel,
|
|
80
|
+
judgeEffort,
|
|
67
81
|
maxTurns: flags.has("--max-turns") ? parseMaxTurns(flags.get("--max-turns")) : void 0
|
|
68
82
|
};
|
|
69
83
|
}
|
|
@@ -72,8 +86,9 @@ function classifyResult(raw) {
|
|
|
72
86
|
const res = root?.results?.results?.[0];
|
|
73
87
|
if (res === void 0) return { error: "promptfoo output carried no result" };
|
|
74
88
|
const message = typeof res.error === "string" ? res.error.trim() : "";
|
|
75
|
-
if (message !== "") return { error: message };
|
|
76
89
|
const stats = root?.results?.stats;
|
|
90
|
+
const gradedByStats = stats !== void 0 && (stats.errors ?? 0) === 0 && ((stats.failures ?? 0) > 0 || (stats.successes ?? 0) > 0);
|
|
91
|
+
if (message !== "" && !gradedByStats) return { error: message };
|
|
77
92
|
if (stats !== void 0 && (stats.errors ?? 0) > 0 && (stats.successes ?? 0) === 0 && (stats.failures ?? 0) === 0) return { error: "promptfoo reported an errored test with nothing graded" };
|
|
78
93
|
if (typeof res.score !== "number" || typeof res.success !== "boolean") return { error: "promptfoo result carried no usable score" };
|
|
79
94
|
return {
|
|
@@ -166,7 +181,7 @@ function discoverScenarios(root) {
|
|
|
166
181
|
}
|
|
167
182
|
function cmdRun(argv) {
|
|
168
183
|
const { positional, flags } = parseArgs(argv);
|
|
169
|
-
if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--harness claude|codex|cursor]");
|
|
184
|
+
if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--judge-effort EFFORT] [--harness claude|codex|cursor]");
|
|
170
185
|
const o = runScenario(positional[0], runOptions(flags), resolveRoot(flags));
|
|
171
186
|
if (o.score === void 0) {
|
|
172
187
|
console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
|
|
@@ -242,7 +257,7 @@ function reduceResults(dir, allowMixed) {
|
|
|
242
257
|
score: res.score,
|
|
243
258
|
pass: res.success,
|
|
244
259
|
agent_model: provider?.config?.model ?? `${harness}-default`,
|
|
245
|
-
judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? "unknown",
|
|
260
|
+
judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? judge?.id ?? "unknown",
|
|
246
261
|
latency_ms: res.latencyMs,
|
|
247
262
|
tokens: (res.tokenUsage?.total ?? 0) + (res.tokenUsage?.assertions?.total ?? 0)
|
|
248
263
|
});
|
package/dist/scenario.js
CHANGED
|
@@ -155,7 +155,10 @@ function buildConfig(s, workdir, manifestPath, opts, paths) {
|
|
|
155
155
|
prompts: ["{{task}}"],
|
|
156
156
|
providers: [agentProvider(opts, workdir, s.skill, paths)],
|
|
157
157
|
defaultTest: { options: {
|
|
158
|
-
provider:
|
|
158
|
+
provider: opts.judgeModel.includes(":") ? opts.judgeEffort === void 0 ? opts.judgeModel : {
|
|
159
|
+
id: opts.judgeModel,
|
|
160
|
+
config: { reasoning_effort: opts.judgeEffort }
|
|
161
|
+
} : process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
|
|
159
162
|
id: "anthropic:claude-agent-sdk",
|
|
160
163
|
config: {
|
|
161
164
|
model: opts.judgeModel,
|
package/docs/usage.md
CHANGED
|
@@ -56,6 +56,20 @@ replays the CLI's `stream-json` output: the `result` event becomes the graded
|
|
|
56
56
|
output and `SKILL.md` reads under `.cursor/skills/` become the `skill-used`
|
|
57
57
|
evidence. the judge leg is unchanged.
|
|
58
58
|
|
|
59
|
+
`--judge` takes either a bare Claude model (graded through the Anthropic
|
|
60
|
+
selection in [auth](#auth)) or a provider-qualified promptfoo id, passed
|
|
61
|
+
through verbatim:
|
|
62
|
+
|
|
63
|
+
```sh
|
|
64
|
+
skillcheck run <scenario-dir> --judge openai:chat:gpt-5.6-sol --judge-effort high
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
a provider-qualified judge authenticates through that provider's own env
|
|
68
|
+
(`OPENAI_API_KEY`, plus `OPENAI_BASE_URL` for a gateway) and is recorded
|
|
69
|
+
verbatim in the scorecard's `judge_model` column. `--judge-effort`
|
|
70
|
+
(minimal|low|medium|high) sets `reasoning_effort` and requires a
|
|
71
|
+
provider-qualified judge; the Anthropic judge does not take one.
|
|
72
|
+
|
|
59
73
|
## sweep
|
|
60
74
|
|
|
61
75
|
```sh
|
|
@@ -130,5 +144,7 @@ written inside the installed package.
|
|
|
130
144
|
| `CODEX_HOME` (default `~/.codex`) | where the codex harness finds the local `codex` CLI login |
|
|
131
145
|
| `OPENAI_API_KEY` | codex agent auth when there is no local login |
|
|
132
146
|
| `CURSOR_API_KEY` | cursor agent auth; a logged-in `cursor-agent` also works |
|
|
147
|
+
| `OPENAI_API_KEY` + `OPENAI_BASE_URL` | a provider-qualified `--judge openai:…`, optionally via a gateway |
|
|
133
148
|
|
|
134
|
-
|
|
149
|
+
a bare `--judge` model stays on the Anthropic selection regardless of the
|
|
150
|
+
agent harness; a provider-qualified `--judge` uses that provider's env instead.
|