@uinaf/skillcheck 0.2.1 → 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/cli.js +10 -7
- package/dist/cursor-provider.js +119 -0
- package/dist/scenario.js +19 -8
- package/docs/releasing.md +4 -4
- package/docs/scenarios.md +9 -8
- package/docs/usage.md +11 -2
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|

|
|
2
2
|
|
|
3
|
-
#
|
|
3
|
+
# uinaf/skillcheck
|
|
4
4
|
|
|
5
5
|
lint and eval harness for agent skills. one CLI with two halves: a keyless
|
|
6
6
|
structural lint any repo can run in CI, and a promptfoo-driven eval loop that
|
package/dist/cli.js
CHANGED
|
@@ -59,7 +59,7 @@ function stateDirs(root) {
|
|
|
59
59
|
}
|
|
60
60
|
function runOptions(flags) {
|
|
61
61
|
const harness = flags.get("--harness") ?? "claude";
|
|
62
|
-
if (harness !== "claude" && harness !== "codex") fail(`--harness must be claude or
|
|
62
|
+
if (harness !== "claude" && harness !== "codex" && harness !== "cursor") fail(`--harness must be claude, codex, or cursor, got ${harness}`);
|
|
63
63
|
return {
|
|
64
64
|
harness,
|
|
65
65
|
agentModel: flags.get("--agent"),
|
|
@@ -72,8 +72,9 @@ function classifyResult(raw) {
|
|
|
72
72
|
const res = root?.results?.results?.[0];
|
|
73
73
|
if (res === void 0) return { error: "promptfoo output carried no result" };
|
|
74
74
|
const message = typeof res.error === "string" ? res.error.trim() : "";
|
|
75
|
-
if (message !== "") return { error: message };
|
|
76
75
|
const stats = root?.results?.stats;
|
|
76
|
+
const gradedByStats = stats !== void 0 && (stats.errors ?? 0) === 0 && ((stats.failures ?? 0) > 0 || (stats.successes ?? 0) > 0);
|
|
77
|
+
if (message !== "" && !gradedByStats) return { error: message };
|
|
77
78
|
if (stats !== void 0 && (stats.errors ?? 0) > 0 && (stats.successes ?? 0) === 0 && (stats.failures ?? 0) === 0) return { error: "promptfoo reported an errored test with nothing graded" };
|
|
78
79
|
if (typeof res.score !== "number" || typeof res.success !== "boolean") return { error: "promptfoo result carried no usable score" };
|
|
79
80
|
return {
|
|
@@ -94,7 +95,8 @@ function runScenario(scenarioDir, opts, root) {
|
|
|
94
95
|
const dirs = stateDirs(root);
|
|
95
96
|
const { name, configPath } = generateRun(path.resolve(scenarioDir), opts, {
|
|
96
97
|
scratchDir: dirs.scratch,
|
|
97
|
-
transformPath: path.join(here, `transform${selfExt}`)
|
|
98
|
+
transformPath: path.join(here, `transform${selfExt}`),
|
|
99
|
+
cursorProviderPath: path.join(here, `cursor-provider${selfExt}`)
|
|
98
100
|
});
|
|
99
101
|
fs.mkdirSync(dirs.results, { recursive: true });
|
|
100
102
|
const resultPath = path.join(dirs.results, `${name}.json`);
|
|
@@ -165,7 +167,7 @@ function discoverScenarios(root) {
|
|
|
165
167
|
}
|
|
166
168
|
function cmdRun(argv) {
|
|
167
169
|
const { positional, flags } = parseArgs(argv);
|
|
168
|
-
if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--harness claude|codex]");
|
|
170
|
+
if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--harness claude|codex|cursor]");
|
|
169
171
|
const o = runScenario(positional[0], runOptions(flags), resolveRoot(flags));
|
|
170
172
|
if (o.score === void 0) {
|
|
171
173
|
console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
|
|
@@ -225,8 +227,9 @@ function reduceResults(dir, allowMixed) {
|
|
|
225
227
|
const provider = raw.config?.providers?.[0];
|
|
226
228
|
const judge = raw.config?.defaultTest?.options?.provider;
|
|
227
229
|
const base = f.replace(/\.json$/, "");
|
|
228
|
-
const
|
|
229
|
-
const
|
|
230
|
+
const suffix = base.match(/--(codex|cursor)$/);
|
|
231
|
+
const harness = suffix === null ? "claude" : suffix[1];
|
|
232
|
+
const [skill, ...rest] = base.replace(/--(codex|cursor)$/, "").split("--");
|
|
230
233
|
let sha = "unattested";
|
|
231
234
|
try {
|
|
232
235
|
sha = JSON.parse(fs.readFileSync(path.join(dir, `${base}.meta.json`), "utf8")).skills_tree_sha ?? "unattested";
|
|
@@ -239,7 +242,7 @@ function reduceResults(dir, allowMixed) {
|
|
|
239
242
|
skills_tree_sha: sha,
|
|
240
243
|
score: res.score,
|
|
241
244
|
pass: res.success,
|
|
242
|
-
agent_model: provider?.config?.model ??
|
|
245
|
+
agent_model: provider?.config?.model ?? `${harness}-default`,
|
|
243
246
|
judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? "unknown",
|
|
244
247
|
latency_ms: res.latencyMs,
|
|
245
248
|
tokens: (res.tokenUsage?.total ?? 0) + (res.tokenUsage?.assertions?.total ?? 0)
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
import { spawn } from "node:child_process";
|
|
2
|
+
//#region src/cursor-provider.ts
|
|
3
|
+
function newStreamState() {
|
|
4
|
+
return {
|
|
5
|
+
isError: false,
|
|
6
|
+
skillCalls: []
|
|
7
|
+
};
|
|
8
|
+
}
|
|
9
|
+
const SKILL_MD = /(?:^|\/)\.cursor\/skills\/([^/]+)\/SKILL\.md$/;
|
|
10
|
+
function foldLine(state, line) {
|
|
11
|
+
let event;
|
|
12
|
+
try {
|
|
13
|
+
event = JSON.parse(line);
|
|
14
|
+
} catch {
|
|
15
|
+
return;
|
|
16
|
+
}
|
|
17
|
+
if (typeof event !== "object" || event === null) return;
|
|
18
|
+
const e = event;
|
|
19
|
+
if (e.type === "tool_call" && e.subtype === "started") {
|
|
20
|
+
const p = e.tool_call?.readToolCall?.args?.path;
|
|
21
|
+
const m = typeof p === "string" ? SKILL_MD.exec(p) : null;
|
|
22
|
+
if (m) state.skillCalls.push({
|
|
23
|
+
name: m[1],
|
|
24
|
+
source: "project",
|
|
25
|
+
path: p
|
|
26
|
+
});
|
|
27
|
+
return;
|
|
28
|
+
}
|
|
29
|
+
if (e.type === "result") {
|
|
30
|
+
state.isError = e.is_error === true || e.subtype !== "success";
|
|
31
|
+
state.result = typeof e.result === "string" ? e.result : "";
|
|
32
|
+
if (e.usage) {
|
|
33
|
+
const prompt = e.usage.inputTokens ?? 0;
|
|
34
|
+
const completion = e.usage.outputTokens ?? 0;
|
|
35
|
+
state.tokenUsage = {
|
|
36
|
+
prompt,
|
|
37
|
+
completion,
|
|
38
|
+
cached: e.usage.cacheReadTokens ?? 0,
|
|
39
|
+
total: prompt + completion
|
|
40
|
+
};
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
var CursorAgentProvider = class {
|
|
45
|
+
providerId;
|
|
46
|
+
config;
|
|
47
|
+
constructor(options) {
|
|
48
|
+
this.providerId = options.id ?? "cursor-agent";
|
|
49
|
+
if (options.config?.working_dir === void 0) throw new Error("cursor-agent provider requires config.working_dir");
|
|
50
|
+
this.config = options.config;
|
|
51
|
+
}
|
|
52
|
+
id() {
|
|
53
|
+
return this.providerId;
|
|
54
|
+
}
|
|
55
|
+
async callApi(prompt) {
|
|
56
|
+
const command = this.config.command ?? "cursor-agent";
|
|
57
|
+
const args = [
|
|
58
|
+
"-p",
|
|
59
|
+
"--trust",
|
|
60
|
+
"--output-format",
|
|
61
|
+
"stream-json"
|
|
62
|
+
];
|
|
63
|
+
if (this.config.model !== void 0) args.push("--model", this.config.model);
|
|
64
|
+
const state = newStreamState();
|
|
65
|
+
const stderr = [];
|
|
66
|
+
let stdoutBuf = "";
|
|
67
|
+
return new Promise((resolve) => {
|
|
68
|
+
const child = spawn(command, args, {
|
|
69
|
+
cwd: this.config.working_dir,
|
|
70
|
+
env: process.env,
|
|
71
|
+
stdio: [
|
|
72
|
+
"pipe",
|
|
73
|
+
"pipe",
|
|
74
|
+
"pipe"
|
|
75
|
+
]
|
|
76
|
+
});
|
|
77
|
+
const timeoutMs = this.config.timeout_ms ?? 9e5;
|
|
78
|
+
const timer = setTimeout(() => {
|
|
79
|
+
child.kill("SIGKILL");
|
|
80
|
+
resolve({ error: `cursor-agent timed out after ${timeoutMs}ms` });
|
|
81
|
+
}, timeoutMs);
|
|
82
|
+
child.on("error", (err) => {
|
|
83
|
+
clearTimeout(timer);
|
|
84
|
+
resolve({ error: `failed to spawn ${command}: ${err.message}` });
|
|
85
|
+
});
|
|
86
|
+
child.stdout.on("data", (chunk) => {
|
|
87
|
+
stdoutBuf += chunk.toString("utf8");
|
|
88
|
+
const lines = stdoutBuf.split("\n");
|
|
89
|
+
stdoutBuf = lines.pop() ?? "";
|
|
90
|
+
for (const line of lines) foldLine(state, line);
|
|
91
|
+
});
|
|
92
|
+
child.stderr.on("data", (chunk) => {
|
|
93
|
+
stderr.push(chunk.toString("utf8"));
|
|
94
|
+
});
|
|
95
|
+
child.on("close", (code) => {
|
|
96
|
+
clearTimeout(timer);
|
|
97
|
+
if (stdoutBuf !== "") foldLine(state, stdoutBuf);
|
|
98
|
+
if (state.result === void 0) {
|
|
99
|
+
const tail = stderr.join("").trim().slice(-2e3);
|
|
100
|
+
resolve({ error: `cursor-agent exited ${code ?? "by signal"} without a result event${tail === "" ? "" : `: ${tail}`}` });
|
|
101
|
+
return;
|
|
102
|
+
}
|
|
103
|
+
if (state.isError) {
|
|
104
|
+
resolve({ error: state.result || "cursor-agent reported an error result" });
|
|
105
|
+
return;
|
|
106
|
+
}
|
|
107
|
+
resolve({
|
|
108
|
+
output: state.result,
|
|
109
|
+
tokenUsage: state.tokenUsage,
|
|
110
|
+
metadata: { skillCalls: state.skillCalls }
|
|
111
|
+
});
|
|
112
|
+
});
|
|
113
|
+
child.stdin.write(prompt);
|
|
114
|
+
child.stdin.end();
|
|
115
|
+
});
|
|
116
|
+
}
|
|
117
|
+
};
|
|
118
|
+
//#endregion
|
|
119
|
+
export { CursorAgentProvider as default, foldLine, newStreamState };
|
package/dist/scenario.js
CHANGED
|
@@ -35,7 +35,7 @@ function loadScenario(scenarioDir) {
|
|
|
35
35
|
function runNameFor(scenarioDir, harness) {
|
|
36
36
|
const m = path.resolve(scenarioDir).match(/skills\/([^/]+)\/evals\/([^/]+)$/);
|
|
37
37
|
if (!m) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
|
|
38
|
-
return harness === "
|
|
38
|
+
return harness === "claude" ? `${m[1]}--${m[2]}` : `${m[1]}--${m[2]}--${harness}`;
|
|
39
39
|
}
|
|
40
40
|
function frontmatterRange(text) {
|
|
41
41
|
const lines = text.split("\n");
|
|
@@ -53,7 +53,11 @@ function stripHiddenFlag(skillMd) {
|
|
|
53
53
|
if (!range) return skillMd;
|
|
54
54
|
return skillMd.split("\n").filter((l, i) => !(i >= range[0] && i < range[1] && l.startsWith("disable-model-invocation:"))).join("\n");
|
|
55
55
|
}
|
|
56
|
-
const RESERVED = /* @__PURE__ */ new Set([
|
|
56
|
+
const RESERVED = /* @__PURE__ */ new Set([
|
|
57
|
+
".claude",
|
|
58
|
+
".agents",
|
|
59
|
+
".cursor"
|
|
60
|
+
]);
|
|
57
61
|
function materialize(s, runDir, harness) {
|
|
58
62
|
const workdir = path.join(runDir, "workdir");
|
|
59
63
|
fs.rmSync(runDir, {
|
|
@@ -78,7 +82,7 @@ function materialize(s, runDir, harness) {
|
|
|
78
82
|
fs.mkdirSync(path.dirname(dest), { recursive: true });
|
|
79
83
|
fs.writeFileSync(dest, content);
|
|
80
84
|
}
|
|
81
|
-
const roots = harness === "codex" ? [".claude", ".agents"] : [".claude"];
|
|
85
|
+
const roots = harness === "codex" ? [".claude", ".agents"] : harness === "cursor" ? [".cursor"] : [".claude"];
|
|
82
86
|
for (const root of roots) fs.cpSync(s.skillDir, path.join(workdir, root, "skills", s.skill), {
|
|
83
87
|
recursive: true,
|
|
84
88
|
filter: (src) => path.basename(src) !== "evals"
|
|
@@ -106,7 +110,14 @@ function materialize(s, runDir, harness) {
|
|
|
106
110
|
manifestPath
|
|
107
111
|
};
|
|
108
112
|
}
|
|
109
|
-
function agentProvider(opts, workdir, skill) {
|
|
113
|
+
function agentProvider(opts, workdir, skill, paths) {
|
|
114
|
+
if (opts.harness === "cursor") return {
|
|
115
|
+
id: `file://${paths.cursorProviderPath}`,
|
|
116
|
+
config: {
|
|
117
|
+
...opts.agentModel ? { model: opts.agentModel } : {},
|
|
118
|
+
working_dir: workdir
|
|
119
|
+
}
|
|
120
|
+
};
|
|
110
121
|
if (opts.harness === "codex") return {
|
|
111
122
|
id: "openai:codex-sdk",
|
|
112
123
|
config: {
|
|
@@ -138,11 +149,11 @@ function agentProvider(opts, workdir, skill) {
|
|
|
138
149
|
}
|
|
139
150
|
};
|
|
140
151
|
}
|
|
141
|
-
function buildConfig(s, workdir, manifestPath, opts,
|
|
152
|
+
function buildConfig(s, workdir, manifestPath, opts, paths) {
|
|
142
153
|
return {
|
|
143
154
|
description: `${s.skill}/${s.scenario}`,
|
|
144
155
|
prompts: ["{{task}}"],
|
|
145
|
-
providers: [agentProvider(opts, workdir, s.skill)],
|
|
156
|
+
providers: [agentProvider(opts, workdir, s.skill, paths)],
|
|
146
157
|
defaultTest: { options: {
|
|
147
158
|
provider: process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
|
|
148
159
|
id: "anthropic:claude-agent-sdk",
|
|
@@ -173,7 +184,7 @@ function buildConfig(s, workdir, manifestPath, opts, transformPath) {
|
|
|
173
184
|
}
|
|
174
185
|
}
|
|
175
186
|
},
|
|
176
|
-
transform: `file://${transformPath}`
|
|
187
|
+
transform: `file://${paths.transformPath}`
|
|
177
188
|
} },
|
|
178
189
|
tests: [{
|
|
179
190
|
description: s.criteria.context,
|
|
@@ -216,7 +227,7 @@ function generateRun(scenarioDir, opts, paths) {
|
|
|
216
227
|
});
|
|
217
228
|
fs.symlinkSync(sdkDir, link, "dir");
|
|
218
229
|
}
|
|
219
|
-
const config = buildConfig(s, workdir, manifestPath, opts, paths
|
|
230
|
+
const config = buildConfig(s, workdir, manifestPath, opts, paths);
|
|
220
231
|
const configPath = path.join(runDir, "promptfooconfig.json");
|
|
221
232
|
fs.writeFileSync(configPath, JSON.stringify(config, null, 2));
|
|
222
233
|
return {
|
package/docs/releasing.md
CHANGED
|
@@ -7,12 +7,12 @@ a push to `main` runs one workflow, `.github/workflows/release.yml`:
|
|
|
7
7
|
```text
|
|
8
8
|
verify ──┐
|
|
9
9
|
├──> release npm publish, OIDC + uinaf-releaser (release environment)
|
|
10
|
-
|
|
10
|
+
scan ────┘
|
|
11
11
|
```
|
|
12
12
|
|
|
13
|
-
`verify` and `
|
|
14
|
-
`
|
|
15
|
-
|
|
13
|
+
`verify` and `scan` are the shared gate: `verify` is called from `verify.yml`,
|
|
14
|
+
and `scan` calls the shared scan in `uinaf/.github`, the same one `scan.yml`
|
|
15
|
+
runs for pull requests. keep it that way: a second copy of the gate on a push-to-`main` workflow
|
|
16
16
|
races this one over the same commit.
|
|
17
17
|
|
|
18
18
|
the file name `release.yml` is load-bearing. see below.
|
package/docs/scenarios.md
CHANGED
|
@@ -8,9 +8,9 @@ a scenario is two files in a frozen location:
|
|
|
8
8
|
```
|
|
9
9
|
|
|
10
10
|
the path is the identity: `<skill>--<scenario>` names the run, the result file,
|
|
11
|
-
and the scorecard entry. on the codex
|
|
12
|
-
so
|
|
13
|
-
is not discovered.
|
|
11
|
+
and the scorecard entry. on the codex and cursor harnesses the name gains a
|
|
12
|
+
`--codex` or `--cursor` suffix, so every harness can hold results side by side.
|
|
13
|
+
a directory missing either file is not discovered.
|
|
14
14
|
|
|
15
15
|
## task.md
|
|
16
16
|
|
|
@@ -27,9 +27,9 @@ Fix the failing check in the config below.
|
|
|
27
27
|
|
|
28
28
|
each block is replaced in the prompt with a pointer ("Input file `config.json`
|
|
29
29
|
is available in your working directory.") and written to disk. destinations
|
|
30
|
-
must stay under the workdir, must not collide, and must not target `.claude
|
|
31
|
-
`.agents/`, since a fixture that writes agent config would be
|
|
32
|
-
own examiner.
|
|
30
|
+
must stay under the workdir, must not collide, and must not target `.claude/`,
|
|
31
|
+
`.agents/`, or `.cursor/`, since a fixture that writes agent config would be
|
|
32
|
+
configuring its own examiner.
|
|
33
33
|
|
|
34
34
|
write the task the way a user would write it. do not name the skill, describe
|
|
35
35
|
its steps, or hint at the checklist: routing is part of what is being measured.
|
|
@@ -81,8 +81,9 @@ frontmatter block; body text mentioning the key does not count.
|
|
|
81
81
|
## the workdir
|
|
82
82
|
|
|
83
83
|
per run, under `<root>/.skillcheck/scratch/<name>/`, rebuilt from scratch each
|
|
84
|
-
time. the skill under test is installed
|
|
85
|
-
`.agents/skills/<skill>/` on
|
|
84
|
+
time. the skill under test is installed where the harness discovers skills —
|
|
85
|
+
`.claude/skills/<skill>/`, plus `.agents/skills/<skill>/` on codex, or
|
|
86
|
+
`.cursor/skills/<skill>/` alone on cursor — with its `evals/` directory
|
|
86
87
|
excluded, so criteria never leak into the agent's context.
|
|
87
88
|
|
|
88
89
|
scenario quality is behavioral proof; [authoring](authoring.md) covers the
|
package/docs/usage.md
CHANGED
|
@@ -45,8 +45,16 @@ message, and writes no provenance sidecar. it is never reported as
|
|
|
45
45
|
`FAIL score=0.0000`; only a real judged verdict can fail a run.
|
|
46
46
|
|
|
47
47
|
defaults: `--harness claude`, agent `claude-opus-5`, judge `claude-opus-5`,
|
|
48
|
-
`--max-turns 50`. on the codex
|
|
49
|
-
the
|
|
48
|
+
`--max-turns 50`. on the codex and cursor harnesses, omitting `--agent` leaves
|
|
49
|
+
the model to that CLI's own default.
|
|
50
|
+
|
|
51
|
+
`--harness cursor` drives the scenario through the Cursor Agent CLI
|
|
52
|
+
(`cursor-agent` on PATH) with the skill installed under `.cursor/skills/`;
|
|
53
|
+
`--agent` names a Cursor model id, e.g. `composer-2.5`. there is no promptfoo
|
|
54
|
+
cursor provider, so the run uses this package's own provider module, which
|
|
55
|
+
replays the CLI's `stream-json` output: the `result` event becomes the graded
|
|
56
|
+
output and `SKILL.md` reads under `.cursor/skills/` become the `skill-used`
|
|
57
|
+
evidence. the judge leg is unchanged.
|
|
50
58
|
|
|
51
59
|
## sweep
|
|
52
60
|
|
|
@@ -121,5 +129,6 @@ written inside the installed package.
|
|
|
121
129
|
| `ANTHROPIC_API_KEY` | judge grades over `anthropic:messages:<model>` instead of the SDK |
|
|
122
130
|
| `CODEX_HOME` (default `~/.codex`) | where the codex harness finds the local `codex` CLI login |
|
|
123
131
|
| `OPENAI_API_KEY` | codex agent auth when there is no local login |
|
|
132
|
+
| `CURSOR_API_KEY` | cursor agent auth; a logged-in `cursor-agent` also works |
|
|
124
133
|
|
|
125
134
|
the judge stays on the Anthropic selection regardless of the agent harness.
|