@uinaf/skillcheck 0.2.1 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -1,6 +1,6 @@
1
1
  ![skillcheck — lint and eval harness for agent skills.](https://uinaf.dev/og/banner/skillcheck.png)
2
2
 
3
- # @uinaf/skillcheck
3
+ # uinaf/skillcheck
4
4
 
5
5
  lint and eval harness for agent skills. one CLI with two halves: a keyless
6
6
  structural lint any repo can run in CI, and a promptfoo-driven eval loop that
package/dist/cli.js CHANGED
@@ -59,7 +59,7 @@ function stateDirs(root) {
59
59
  }
60
60
  function runOptions(flags) {
61
61
  const harness = flags.get("--harness") ?? "claude";
62
- if (harness !== "claude" && harness !== "codex") fail(`--harness must be claude or codex, got ${harness}`);
62
+ if (harness !== "claude" && harness !== "codex" && harness !== "cursor") fail(`--harness must be claude, codex, or cursor, got ${harness}`);
63
63
  return {
64
64
  harness,
65
65
  agentModel: flags.get("--agent"),
@@ -94,7 +94,8 @@ function runScenario(scenarioDir, opts, root) {
94
94
  const dirs = stateDirs(root);
95
95
  const { name, configPath } = generateRun(path.resolve(scenarioDir), opts, {
96
96
  scratchDir: dirs.scratch,
97
- transformPath: path.join(here, `transform${selfExt}`)
97
+ transformPath: path.join(here, `transform${selfExt}`),
98
+ cursorProviderPath: path.join(here, `cursor-provider${selfExt}`)
98
99
  });
99
100
  fs.mkdirSync(dirs.results, { recursive: true });
100
101
  const resultPath = path.join(dirs.results, `${name}.json`);
@@ -165,7 +166,7 @@ function discoverScenarios(root) {
165
166
  }
166
167
  function cmdRun(argv) {
167
168
  const { positional, flags } = parseArgs(argv);
168
- if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--harness claude|codex]");
169
+ if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--harness claude|codex|cursor]");
169
170
  const o = runScenario(positional[0], runOptions(flags), resolveRoot(flags));
170
171
  if (o.score === void 0) {
171
172
  console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
@@ -225,8 +226,9 @@ function reduceResults(dir, allowMixed) {
225
226
  const provider = raw.config?.providers?.[0];
226
227
  const judge = raw.config?.defaultTest?.options?.provider;
227
228
  const base = f.replace(/\.json$/, "");
228
- const harness = base.endsWith("--codex") ? "codex" : "claude";
229
- const [skill, ...rest] = base.replace(/--codex$/, "").split("--");
229
+ const suffix = base.match(/--(codex|cursor)$/);
230
+ const harness = suffix === null ? "claude" : suffix[1];
231
+ const [skill, ...rest] = base.replace(/--(codex|cursor)$/, "").split("--");
230
232
  let sha = "unattested";
231
233
  try {
232
234
  sha = JSON.parse(fs.readFileSync(path.join(dir, `${base}.meta.json`), "utf8")).skills_tree_sha ?? "unattested";
@@ -239,7 +241,7 @@ function reduceResults(dir, allowMixed) {
239
241
  skills_tree_sha: sha,
240
242
  score: res.score,
241
243
  pass: res.success,
242
- agent_model: provider?.config?.model ?? "codex-default",
244
+ agent_model: provider?.config?.model ?? `${harness}-default`,
243
245
  judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? "unknown",
244
246
  latency_ms: res.latencyMs,
245
247
  tokens: (res.tokenUsage?.total ?? 0) + (res.tokenUsage?.assertions?.total ?? 0)
@@ -0,0 +1,119 @@
1
+ import { spawn } from "node:child_process";
2
+ //#region src/cursor-provider.ts
3
+ function newStreamState() {
4
+ return {
5
+ isError: false,
6
+ skillCalls: []
7
+ };
8
+ }
9
+ const SKILL_MD = /(?:^|\/)\.cursor\/skills\/([^/]+)\/SKILL\.md$/;
10
+ function foldLine(state, line) {
11
+ let event;
12
+ try {
13
+ event = JSON.parse(line);
14
+ } catch {
15
+ return;
16
+ }
17
+ if (typeof event !== "object" || event === null) return;
18
+ const e = event;
19
+ if (e.type === "tool_call" && e.subtype === "started") {
20
+ const p = e.tool_call?.readToolCall?.args?.path;
21
+ const m = typeof p === "string" ? SKILL_MD.exec(p) : null;
22
+ if (m) state.skillCalls.push({
23
+ name: m[1],
24
+ source: "project",
25
+ path: p
26
+ });
27
+ return;
28
+ }
29
+ if (e.type === "result") {
30
+ state.isError = e.is_error === true || e.subtype !== "success";
31
+ state.result = typeof e.result === "string" ? e.result : "";
32
+ if (e.usage) {
33
+ const prompt = e.usage.inputTokens ?? 0;
34
+ const completion = e.usage.outputTokens ?? 0;
35
+ state.tokenUsage = {
36
+ prompt,
37
+ completion,
38
+ cached: e.usage.cacheReadTokens ?? 0,
39
+ total: prompt + completion
40
+ };
41
+ }
42
+ }
43
+ }
44
+ var CursorAgentProvider = class {
45
+ providerId;
46
+ config;
47
+ constructor(options) {
48
+ this.providerId = options.id ?? "cursor-agent";
49
+ if (options.config?.working_dir === void 0) throw new Error("cursor-agent provider requires config.working_dir");
50
+ this.config = options.config;
51
+ }
52
+ id() {
53
+ return this.providerId;
54
+ }
55
+ async callApi(prompt) {
56
+ const command = this.config.command ?? "cursor-agent";
57
+ const args = [
58
+ "-p",
59
+ "--trust",
60
+ "--output-format",
61
+ "stream-json"
62
+ ];
63
+ if (this.config.model !== void 0) args.push("--model", this.config.model);
64
+ const state = newStreamState();
65
+ const stderr = [];
66
+ let stdoutBuf = "";
67
+ return new Promise((resolve) => {
68
+ const child = spawn(command, args, {
69
+ cwd: this.config.working_dir,
70
+ env: process.env,
71
+ stdio: [
72
+ "pipe",
73
+ "pipe",
74
+ "pipe"
75
+ ]
76
+ });
77
+ const timeoutMs = this.config.timeout_ms ?? 9e5;
78
+ const timer = setTimeout(() => {
79
+ child.kill("SIGKILL");
80
+ resolve({ error: `cursor-agent timed out after ${timeoutMs}ms` });
81
+ }, timeoutMs);
82
+ child.on("error", (err) => {
83
+ clearTimeout(timer);
84
+ resolve({ error: `failed to spawn ${command}: ${err.message}` });
85
+ });
86
+ child.stdout.on("data", (chunk) => {
87
+ stdoutBuf += chunk.toString("utf8");
88
+ const lines = stdoutBuf.split("\n");
89
+ stdoutBuf = lines.pop() ?? "";
90
+ for (const line of lines) foldLine(state, line);
91
+ });
92
+ child.stderr.on("data", (chunk) => {
93
+ stderr.push(chunk.toString("utf8"));
94
+ });
95
+ child.on("close", (code) => {
96
+ clearTimeout(timer);
97
+ if (stdoutBuf !== "") foldLine(state, stdoutBuf);
98
+ if (state.result === void 0) {
99
+ const tail = stderr.join("").trim().slice(-2e3);
100
+ resolve({ error: `cursor-agent exited ${code ?? "by signal"} without a result event${tail === "" ? "" : `: ${tail}`}` });
101
+ return;
102
+ }
103
+ if (state.isError) {
104
+ resolve({ error: state.result || "cursor-agent reported an error result" });
105
+ return;
106
+ }
107
+ resolve({
108
+ output: state.result,
109
+ tokenUsage: state.tokenUsage,
110
+ metadata: { skillCalls: state.skillCalls }
111
+ });
112
+ });
113
+ child.stdin.write(prompt);
114
+ child.stdin.end();
115
+ });
116
+ }
117
+ };
118
+ //#endregion
119
+ export { CursorAgentProvider as default, foldLine, newStreamState };
package/dist/scenario.js CHANGED
@@ -35,7 +35,7 @@ function loadScenario(scenarioDir) {
35
35
  function runNameFor(scenarioDir, harness) {
36
36
  const m = path.resolve(scenarioDir).match(/skills\/([^/]+)\/evals\/([^/]+)$/);
37
37
  if (!m) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
38
- return harness === "codex" ? `${m[1]}--${m[2]}--codex` : `${m[1]}--${m[2]}`;
38
+ return harness === "claude" ? `${m[1]}--${m[2]}` : `${m[1]}--${m[2]}--${harness}`;
39
39
  }
40
40
  function frontmatterRange(text) {
41
41
  const lines = text.split("\n");
@@ -53,7 +53,11 @@ function stripHiddenFlag(skillMd) {
53
53
  if (!range) return skillMd;
54
54
  return skillMd.split("\n").filter((l, i) => !(i >= range[0] && i < range[1] && l.startsWith("disable-model-invocation:"))).join("\n");
55
55
  }
56
- const RESERVED = /* @__PURE__ */ new Set([".claude", ".agents"]);
56
+ const RESERVED = /* @__PURE__ */ new Set([
57
+ ".claude",
58
+ ".agents",
59
+ ".cursor"
60
+ ]);
57
61
  function materialize(s, runDir, harness) {
58
62
  const workdir = path.join(runDir, "workdir");
59
63
  fs.rmSync(runDir, {
@@ -78,7 +82,7 @@ function materialize(s, runDir, harness) {
78
82
  fs.mkdirSync(path.dirname(dest), { recursive: true });
79
83
  fs.writeFileSync(dest, content);
80
84
  }
81
- const roots = harness === "codex" ? [".claude", ".agents"] : [".claude"];
85
+ const roots = harness === "codex" ? [".claude", ".agents"] : harness === "cursor" ? [".cursor"] : [".claude"];
82
86
  for (const root of roots) fs.cpSync(s.skillDir, path.join(workdir, root, "skills", s.skill), {
83
87
  recursive: true,
84
88
  filter: (src) => path.basename(src) !== "evals"
@@ -106,7 +110,14 @@ function materialize(s, runDir, harness) {
106
110
  manifestPath
107
111
  };
108
112
  }
109
- function agentProvider(opts, workdir, skill) {
113
+ function agentProvider(opts, workdir, skill, paths) {
114
+ if (opts.harness === "cursor") return {
115
+ id: `file://${paths.cursorProviderPath}`,
116
+ config: {
117
+ ...opts.agentModel ? { model: opts.agentModel } : {},
118
+ working_dir: workdir
119
+ }
120
+ };
110
121
  if (opts.harness === "codex") return {
111
122
  id: "openai:codex-sdk",
112
123
  config: {
@@ -138,11 +149,11 @@ function agentProvider(opts, workdir, skill) {
138
149
  }
139
150
  };
140
151
  }
141
- function buildConfig(s, workdir, manifestPath, opts, transformPath) {
152
+ function buildConfig(s, workdir, manifestPath, opts, paths) {
142
153
  return {
143
154
  description: `${s.skill}/${s.scenario}`,
144
155
  prompts: ["{{task}}"],
145
- providers: [agentProvider(opts, workdir, s.skill)],
156
+ providers: [agentProvider(opts, workdir, s.skill, paths)],
146
157
  defaultTest: { options: {
147
158
  provider: process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
148
159
  id: "anthropic:claude-agent-sdk",
@@ -173,7 +184,7 @@ function buildConfig(s, workdir, manifestPath, opts, transformPath) {
173
184
  }
174
185
  }
175
186
  },
176
- transform: `file://${transformPath}`
187
+ transform: `file://${paths.transformPath}`
177
188
  } },
178
189
  tests: [{
179
190
  description: s.criteria.context,
@@ -216,7 +227,7 @@ function generateRun(scenarioDir, opts, paths) {
216
227
  });
217
228
  fs.symlinkSync(sdkDir, link, "dir");
218
229
  }
219
- const config = buildConfig(s, workdir, manifestPath, opts, paths.transformPath);
230
+ const config = buildConfig(s, workdir, manifestPath, opts, paths);
220
231
  const configPath = path.join(runDir, "promptfooconfig.json");
221
232
  fs.writeFileSync(configPath, JSON.stringify(config, null, 2));
222
233
  return {
package/docs/releasing.md CHANGED
@@ -7,12 +7,12 @@ a push to `main` runs one workflow, `.github/workflows/release.yml`:
7
7
  ```text
8
8
  verify ──┐
9
9
  ├──> release npm publish, OIDC + uinaf-releaser (release environment)
10
- secrets ─┘
10
+ scan ────┘
11
11
  ```
12
12
 
13
- `verify` and `secrets` are the shared gate, called from `verify.yml` and
14
- `secrets.yml`, so one definition serves pull requests, the merge queue, and this
15
- push. keep it that way: a second copy of the gate on a push-to-`main` workflow
13
+ `verify` and `scan` are the shared gate: `verify` is called from `verify.yml`,
14
+ and `scan` calls the shared scan in `uinaf/.github`, the same one `scan.yml`
15
+ runs for pull requests. keep it that way: a second copy of the gate on a push-to-`main` workflow
16
16
  races this one over the same commit.
17
17
 
18
18
  the file name `release.yml` is load-bearing. see below.
package/docs/scenarios.md CHANGED
@@ -8,9 +8,9 @@ a scenario is two files in a frozen location:
8
8
  ```
9
9
 
10
10
  the path is the identity: `<skill>--<scenario>` names the run, the result file,
11
- and the scorecard entry. on the codex harness the name gains a `--codex` suffix,
12
- so both harnesses can hold results side by side. a directory missing either file
13
- is not discovered.
11
+ and the scorecard entry. on the codex and cursor harnesses the name gains a
12
+ `--codex` or `--cursor` suffix, so every harness can hold results side by side.
13
+ a directory missing either file is not discovered.
14
14
 
15
15
  ## task.md
16
16
 
@@ -27,9 +27,9 @@ Fix the failing check in the config below.
27
27
 
28
28
  each block is replaced in the prompt with a pointer ("Input file `config.json`
29
29
  is available in your working directory.") and written to disk. destinations
30
- must stay under the workdir, must not collide, and must not target `.claude/` or
31
- `.agents/`, since a fixture that writes agent config would be configuring its
32
- own examiner.
30
+ must stay under the workdir, must not collide, and must not target `.claude/`,
31
+ `.agents/`, or `.cursor/`, since a fixture that writes agent config would be
32
+ configuring its own examiner.
33
33
 
34
34
  write the task the way a user would write it. do not name the skill, describe
35
35
  its steps, or hint at the checklist: routing is part of what is being measured.
@@ -81,8 +81,9 @@ frontmatter block; body text mentioning the key does not count.
81
81
  ## the workdir
82
82
 
83
83
  per run, under `<root>/.skillcheck/scratch/<name>/`, rebuilt from scratch each
84
- time. the skill under test is installed at `.claude/skills/<skill>/` (and also
85
- `.agents/skills/<skill>/` on the codex harness) with its `evals/` directory
84
+ time. the skill under test is installed where the harness discovers skills —
85
+ `.claude/skills/<skill>/`, plus `.agents/skills/<skill>/` on codex, or
86
+ `.cursor/skills/<skill>/` alone on cursor — with its `evals/` directory
86
87
  excluded, so criteria never leak into the agent's context.
87
88
 
88
89
  scenario quality is behavioral proof; [authoring](authoring.md) covers the
package/docs/usage.md CHANGED
@@ -45,8 +45,16 @@ message, and writes no provenance sidecar. it is never reported as
45
45
  `FAIL score=0.0000`; only a real judged verdict can fail a run.
46
46
 
47
47
  defaults: `--harness claude`, agent `claude-opus-5`, judge `claude-opus-5`,
48
- `--max-turns 50`. on the codex harness, omitting `--agent` leaves the model to
49
- the Codex CLI's own default.
48
+ `--max-turns 50`. on the codex and cursor harnesses, omitting `--agent` leaves
49
+ the model to that CLI's own default.
50
+
51
+ `--harness cursor` drives the scenario through the Cursor Agent CLI
52
+ (`cursor-agent` on PATH) with the skill installed under `.cursor/skills/`;
53
+ `--agent` names a Cursor model id, e.g. `composer-2.5`. there is no promptfoo
54
+ cursor provider, so the run uses this package's own provider module, which
55
+ replays the CLI's `stream-json` output: the `result` event becomes the graded
56
+ output and `SKILL.md` reads under `.cursor/skills/` become the `skill-used`
57
+ evidence. the judge leg is unchanged.
50
58
 
51
59
  ## sweep
52
60
 
@@ -121,5 +129,6 @@ written inside the installed package.
121
129
  | `ANTHROPIC_API_KEY` | judge grades over `anthropic:messages:<model>` instead of the SDK |
122
130
  | `CODEX_HOME` (default `~/.codex`) | where the codex harness finds the local `codex` CLI login |
123
131
  | `OPENAI_API_KEY` | codex agent auth when there is no local login |
132
+ | `CURSOR_API_KEY` | cursor agent auth; a logged-in `cursor-agent` also works |
124
133
 
125
134
  the judge stays on the Anthropic selection regardless of the agent harness.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "0.2.1",
3
+ "version": "0.3.0",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {