@uinaf/skillcheck 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -1,6 +1,6 @@
1
1
  ![skillcheck — lint and eval harness for agent skills.](https://uinaf.dev/og/banner/skillcheck.png)
2
2
 
3
- # @uinaf/skillcheck
3
+ # uinaf/skillcheck
4
4
 
5
5
  lint and eval harness for agent skills. one CLI with two halves: a keyless
6
6
  structural lint any repo can run in CI, and a promptfoo-driven eval loop that
@@ -49,14 +49,15 @@ CLI it documents.
49
49
 
50
50
  ## docs
51
51
 
52
- | doc | when |
53
- | ------------------------------- | ------------------------------------- |
54
- | [usage](docs/usage.md) | every subcommand, flag, and auth path |
55
- | [scenarios](docs/scenarios.md) | writing an eval scenario |
56
- | [adoption](docs/adoption.md) | wiring the lint into another repo |
57
- | [releasing](docs/releasing.md) | the npm pipeline |
58
- | [contributing](CONTRIBUTING.md) | local setup and the verify gate |
59
- | [security](SECURITY.md) | reporting a vulnerability |
52
+ | doc | when |
53
+ | --------------------------------------------------------------- | ------------------------------------- |
54
+ | [usage](docs/usage.md) | every subcommand, flag, and auth path |
55
+ | [scenarios](docs/scenarios.md) | writing an eval scenario |
56
+ | [authoring](docs/authoring.md) | writing and auditing the skill itself |
57
+ | [adoption](docs/adoption.md) | wiring the lint into another repo |
58
+ | [releasing](docs/releasing.md) | the npm pipeline |
59
+ | [contributing](CONTRIBUTING.md) | local setup and the verify gate |
60
+ | [security](https://github.com/uinaf/skillcheck/security/policy) | reporting a vulnerability |
60
61
 
61
62
  ## license
62
63
 
package/dist/cli.js CHANGED
@@ -59,7 +59,7 @@ function stateDirs(root) {
59
59
  }
60
60
  function runOptions(flags) {
61
61
  const harness = flags.get("--harness") ?? "claude";
62
- if (harness !== "claude" && harness !== "codex") fail(`--harness must be claude or codex, got ${harness}`);
62
+ if (harness !== "claude" && harness !== "codex" && harness !== "cursor") fail(`--harness must be claude, codex, or cursor, got ${harness}`);
63
63
  return {
64
64
  harness,
65
65
  agentModel: flags.get("--agent"),
@@ -94,7 +94,8 @@ function runScenario(scenarioDir, opts, root) {
94
94
  const dirs = stateDirs(root);
95
95
  const { name, configPath } = generateRun(path.resolve(scenarioDir), opts, {
96
96
  scratchDir: dirs.scratch,
97
- transformPath: path.join(here, `transform${selfExt}`)
97
+ transformPath: path.join(here, `transform${selfExt}`),
98
+ cursorProviderPath: path.join(here, `cursor-provider${selfExt}`)
98
99
  });
99
100
  fs.mkdirSync(dirs.results, { recursive: true });
100
101
  const resultPath = path.join(dirs.results, `${name}.json`);
@@ -165,7 +166,7 @@ function discoverScenarios(root) {
165
166
  }
166
167
  function cmdRun(argv) {
167
168
  const { positional, flags } = parseArgs(argv);
168
- if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--harness claude|codex]");
169
+ if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--harness claude|codex|cursor]");
169
170
  const o = runScenario(positional[0], runOptions(flags), resolveRoot(flags));
170
171
  if (o.score === void 0) {
171
172
  console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
@@ -225,8 +226,9 @@ function reduceResults(dir, allowMixed) {
225
226
  const provider = raw.config?.providers?.[0];
226
227
  const judge = raw.config?.defaultTest?.options?.provider;
227
228
  const base = f.replace(/\.json$/, "");
228
- const harness = base.endsWith("--codex") ? "codex" : "claude";
229
- const [skill, ...rest] = base.replace(/--codex$/, "").split("--");
229
+ const suffix = base.match(/--(codex|cursor)$/);
230
+ const harness = suffix === null ? "claude" : suffix[1];
231
+ const [skill, ...rest] = base.replace(/--(codex|cursor)$/, "").split("--");
230
232
  let sha = "unattested";
231
233
  try {
232
234
  sha = JSON.parse(fs.readFileSync(path.join(dir, `${base}.meta.json`), "utf8")).skills_tree_sha ?? "unattested";
@@ -239,7 +241,7 @@ function reduceResults(dir, allowMixed) {
239
241
  skills_tree_sha: sha,
240
242
  score: res.score,
241
243
  pass: res.success,
242
- agent_model: provider?.config?.model ?? "codex-default",
244
+ agent_model: provider?.config?.model ?? `${harness}-default`,
243
245
  judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? "unknown",
244
246
  latency_ms: res.latencyMs,
245
247
  tokens: (res.tokenUsage?.total ?? 0) + (res.tokenUsage?.assertions?.total ?? 0)
@@ -0,0 +1,119 @@
1
+ import { spawn } from "node:child_process";
2
+ //#region src/cursor-provider.ts
3
+ function newStreamState() {
4
+ return {
5
+ isError: false,
6
+ skillCalls: []
7
+ };
8
+ }
9
+ const SKILL_MD = /(?:^|\/)\.cursor\/skills\/([^/]+)\/SKILL\.md$/;
10
+ function foldLine(state, line) {
11
+ let event;
12
+ try {
13
+ event = JSON.parse(line);
14
+ } catch {
15
+ return;
16
+ }
17
+ if (typeof event !== "object" || event === null) return;
18
+ const e = event;
19
+ if (e.type === "tool_call" && e.subtype === "started") {
20
+ const p = e.tool_call?.readToolCall?.args?.path;
21
+ const m = typeof p === "string" ? SKILL_MD.exec(p) : null;
22
+ if (m) state.skillCalls.push({
23
+ name: m[1],
24
+ source: "project",
25
+ path: p
26
+ });
27
+ return;
28
+ }
29
+ if (e.type === "result") {
30
+ state.isError = e.is_error === true || e.subtype !== "success";
31
+ state.result = typeof e.result === "string" ? e.result : "";
32
+ if (e.usage) {
33
+ const prompt = e.usage.inputTokens ?? 0;
34
+ const completion = e.usage.outputTokens ?? 0;
35
+ state.tokenUsage = {
36
+ prompt,
37
+ completion,
38
+ cached: e.usage.cacheReadTokens ?? 0,
39
+ total: prompt + completion
40
+ };
41
+ }
42
+ }
43
+ }
44
+ var CursorAgentProvider = class {
45
+ providerId;
46
+ config;
47
+ constructor(options) {
48
+ this.providerId = options.id ?? "cursor-agent";
49
+ if (options.config?.working_dir === void 0) throw new Error("cursor-agent provider requires config.working_dir");
50
+ this.config = options.config;
51
+ }
52
+ id() {
53
+ return this.providerId;
54
+ }
55
+ async callApi(prompt) {
56
+ const command = this.config.command ?? "cursor-agent";
57
+ const args = [
58
+ "-p",
59
+ "--trust",
60
+ "--output-format",
61
+ "stream-json"
62
+ ];
63
+ if (this.config.model !== void 0) args.push("--model", this.config.model);
64
+ const state = newStreamState();
65
+ const stderr = [];
66
+ let stdoutBuf = "";
67
+ return new Promise((resolve) => {
68
+ const child = spawn(command, args, {
69
+ cwd: this.config.working_dir,
70
+ env: process.env,
71
+ stdio: [
72
+ "pipe",
73
+ "pipe",
74
+ "pipe"
75
+ ]
76
+ });
77
+ const timeoutMs = this.config.timeout_ms ?? 9e5;
78
+ const timer = setTimeout(() => {
79
+ child.kill("SIGKILL");
80
+ resolve({ error: `cursor-agent timed out after ${timeoutMs}ms` });
81
+ }, timeoutMs);
82
+ child.on("error", (err) => {
83
+ clearTimeout(timer);
84
+ resolve({ error: `failed to spawn ${command}: ${err.message}` });
85
+ });
86
+ child.stdout.on("data", (chunk) => {
87
+ stdoutBuf += chunk.toString("utf8");
88
+ const lines = stdoutBuf.split("\n");
89
+ stdoutBuf = lines.pop() ?? "";
90
+ for (const line of lines) foldLine(state, line);
91
+ });
92
+ child.stderr.on("data", (chunk) => {
93
+ stderr.push(chunk.toString("utf8"));
94
+ });
95
+ child.on("close", (code) => {
96
+ clearTimeout(timer);
97
+ if (stdoutBuf !== "") foldLine(state, stdoutBuf);
98
+ if (state.result === void 0) {
99
+ const tail = stderr.join("").trim().slice(-2e3);
100
+ resolve({ error: `cursor-agent exited ${code ?? "by signal"} without a result event${tail === "" ? "" : `: ${tail}`}` });
101
+ return;
102
+ }
103
+ if (state.isError) {
104
+ resolve({ error: state.result || "cursor-agent reported an error result" });
105
+ return;
106
+ }
107
+ resolve({
108
+ output: state.result,
109
+ tokenUsage: state.tokenUsage,
110
+ metadata: { skillCalls: state.skillCalls }
111
+ });
112
+ });
113
+ child.stdin.write(prompt);
114
+ child.stdin.end();
115
+ });
116
+ }
117
+ };
118
+ //#endregion
119
+ export { CursorAgentProvider as default, foldLine, newStreamState };
package/dist/scenario.js CHANGED
@@ -35,7 +35,7 @@ function loadScenario(scenarioDir) {
35
35
  function runNameFor(scenarioDir, harness) {
36
36
  const m = path.resolve(scenarioDir).match(/skills\/([^/]+)\/evals\/([^/]+)$/);
37
37
  if (!m) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
38
- return harness === "codex" ? `${m[1]}--${m[2]}--codex` : `${m[1]}--${m[2]}`;
38
+ return harness === "claude" ? `${m[1]}--${m[2]}` : `${m[1]}--${m[2]}--${harness}`;
39
39
  }
40
40
  function frontmatterRange(text) {
41
41
  const lines = text.split("\n");
@@ -53,7 +53,11 @@ function stripHiddenFlag(skillMd) {
53
53
  if (!range) return skillMd;
54
54
  return skillMd.split("\n").filter((l, i) => !(i >= range[0] && i < range[1] && l.startsWith("disable-model-invocation:"))).join("\n");
55
55
  }
56
- const RESERVED = /* @__PURE__ */ new Set([".claude", ".agents"]);
56
+ const RESERVED = /* @__PURE__ */ new Set([
57
+ ".claude",
58
+ ".agents",
59
+ ".cursor"
60
+ ]);
57
61
  function materialize(s, runDir, harness) {
58
62
  const workdir = path.join(runDir, "workdir");
59
63
  fs.rmSync(runDir, {
@@ -78,7 +82,7 @@ function materialize(s, runDir, harness) {
78
82
  fs.mkdirSync(path.dirname(dest), { recursive: true });
79
83
  fs.writeFileSync(dest, content);
80
84
  }
81
- const roots = harness === "codex" ? [".claude", ".agents"] : [".claude"];
85
+ const roots = harness === "codex" ? [".claude", ".agents"] : harness === "cursor" ? [".cursor"] : [".claude"];
82
86
  for (const root of roots) fs.cpSync(s.skillDir, path.join(workdir, root, "skills", s.skill), {
83
87
  recursive: true,
84
88
  filter: (src) => path.basename(src) !== "evals"
@@ -106,7 +110,14 @@ function materialize(s, runDir, harness) {
106
110
  manifestPath
107
111
  };
108
112
  }
109
- function agentProvider(opts, workdir, skill) {
113
+ function agentProvider(opts, workdir, skill, paths) {
114
+ if (opts.harness === "cursor") return {
115
+ id: `file://${paths.cursorProviderPath}`,
116
+ config: {
117
+ ...opts.agentModel ? { model: opts.agentModel } : {},
118
+ working_dir: workdir
119
+ }
120
+ };
110
121
  if (opts.harness === "codex") return {
111
122
  id: "openai:codex-sdk",
112
123
  config: {
@@ -138,11 +149,11 @@ function agentProvider(opts, workdir, skill) {
138
149
  }
139
150
  };
140
151
  }
141
- function buildConfig(s, workdir, manifestPath, opts, transformPath) {
152
+ function buildConfig(s, workdir, manifestPath, opts, paths) {
142
153
  return {
143
154
  description: `${s.skill}/${s.scenario}`,
144
155
  prompts: ["{{task}}"],
145
- providers: [agentProvider(opts, workdir, s.skill)],
156
+ providers: [agentProvider(opts, workdir, s.skill, paths)],
146
157
  defaultTest: { options: {
147
158
  provider: process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
148
159
  id: "anthropic:claude-agent-sdk",
@@ -173,7 +184,7 @@ function buildConfig(s, workdir, manifestPath, opts, transformPath) {
173
184
  }
174
185
  }
175
186
  },
176
- transform: `file://${transformPath}`
187
+ transform: `file://${paths.transformPath}`
177
188
  } },
178
189
  tests: [{
179
190
  description: s.criteria.context,
@@ -216,7 +227,7 @@ function generateRun(scenarioDir, opts, paths) {
216
227
  });
217
228
  fs.symlinkSync(sdkDir, link, "dir");
218
229
  }
219
- const config = buildConfig(s, workdir, manifestPath, opts, paths.transformPath);
230
+ const config = buildConfig(s, workdir, manifestPath, opts, paths);
220
231
  const configPath = path.join(runDir, "promptfooconfig.json");
221
232
  fs.writeFileSync(configPath, JSON.stringify(config, null, 2));
222
233
  return {
@@ -0,0 +1,70 @@
1
+ # authoring and auditing skills
2
+
3
+ what `lint` and evals cannot judge: whether a skill is worth routing to and
4
+ cheap to load. use this when writing a skill or auditing one. evidence beats
5
+ stylistic preference; run `skillcheck lint` first and let this cover the rest.
6
+
7
+ ## metadata and discovery
8
+
9
+ - `name` is concrete and easy to say out loud. `helper`, `tools`, `utils` are
10
+ discovery smells.
11
+ - `description` is third person and says both what the skill does and when to
12
+ use it. it is an always-loaded retrieval pointer: front-load the concrete
13
+ action or domain that should activate it.
14
+ - one trigger per materially distinct request branch. collapse synonyms that
15
+ rename the same branch.
16
+ - state the main overlap boundary without naming another skill.
17
+
18
+ ## body shape
19
+
20
+ - keep `SKILL.md` on workflow, principles, boundaries, and routing. lead with
21
+ the task, not a bibliography.
22
+ - assume the model is smart; spend tokens on repo-specific judgment. delete any
23
+ instruction that would not change a capable model's behavior.
24
+ - match freedom to risk: high for contextual judgment, medium when a preferred
25
+ pattern exists, low for fragile operations.
26
+ - say what evidence to gather and what a complete result includes. end each
27
+ step with an observable completion condition, not "understood" or "handled".
28
+
29
+ ## progressive disclosure
30
+
31
+ - durable detail, rubrics, and long examples go in `references/`, one hop from
32
+ `SKILL.md`, each with a task-shaped retrieval job. material every path needs
33
+ stays inline.
34
+ - for repeated deterministic work, route to the target's existing framework,
35
+ schema, task graph, or library; otherwise add a tested module in the
36
+ project's primary language, not ad-hoc shell rendered as prose.
37
+ - when executable code belongs to another maintained project, link the exact
38
+ public artifact and state the contract it demonstrates; do not fork it into
39
+ prose.
40
+ - a package stays independently usable: state prerequisites and out-of-scope
41
+ next steps locally. never invoke, import, or assume a sibling skill.
42
+
43
+ ## audit
44
+
45
+ grade each dimension strong, mixed, or weak:
46
+
47
+ | dimension | question |
48
+ | ---------------------- | --------------------------------------------------------------------- |
49
+ | discovery | does metadata alone route a realistic request here |
50
+ | workflow | does the body say how to begin, what evidence to gather, when to stop |
51
+ | progressive disclosure | is detail in the right file |
52
+ | repo fit | are links, commands, and conventions current |
53
+ | verification | is the strongest mechanical check named, plus a real evidence loop |
54
+ | boundaries | are limits and next steps stated without leaning on a sibling skill |
55
+
56
+ blockers, must-fix: invalid frontmatter; a description that fails discovery;
57
+ stale commands, paths, or links; a workflow with no start, evidence loop, or
58
+ completion; conflicts with the repo's guidance; sibling-skill dependencies.
59
+
60
+ major findings: vague name; synonym-stuffed description; bloated `SKILL.md`;
61
+ missing or muddy boundaries; prose re-inventing a deterministic tool; abstract
62
+ examples.
63
+
64
+ ## improve
65
+
66
+ fix blockers first, then the highest-leverage majors. prefer the smallest
67
+ change that improves activation, decision quality, or proof. when pruning,
68
+ measure common-path context for representative requests; line count alone does
69
+ not reveal retrieval cost. after edits, rerun `skillcheck lint` and the repo's
70
+ gate, and rerun evals when behavior was the thing changed.
package/docs/releasing.md CHANGED
@@ -7,12 +7,12 @@ a push to `main` runs one workflow, `.github/workflows/release.yml`:
7
7
  ```text
8
8
  verify ──┐
9
9
  ├──> release npm publish, OIDC + uinaf-releaser (release environment)
10
- secrets ─┘
10
+ scan ────┘
11
11
  ```
12
12
 
13
- `verify` and `secrets` are the shared gate, called from `verify.yml` and
14
- `secrets.yml`, so one definition serves pull requests, the merge queue, and this
15
- push. keep it that way: a second copy of the gate on a push-to-`main` workflow
13
+ `verify` and `scan` are the shared gate: `verify` is called from `verify.yml`,
14
+ and `scan` calls the shared scan in `uinaf/.github`, the same one `scan.yml`
15
+ runs for pull requests. keep it that way: a second copy of the gate on a push-to-`main` workflow
16
16
  races this one over the same commit.
17
17
 
18
18
  the file name `release.yml` is load-bearing. see below.
package/docs/scenarios.md CHANGED
@@ -8,9 +8,9 @@ a scenario is two files in a frozen location:
8
8
  ```
9
9
 
10
10
  the path is the identity: `<skill>--<scenario>` names the run, the result file,
11
- and the scorecard entry. on the codex harness the name gains a `--codex` suffix,
12
- so both harnesses can hold results side by side. a directory missing either file
13
- is not discovered.
11
+ and the scorecard entry. on the codex and cursor harnesses the name gains a
12
+ `--codex` or `--cursor` suffix, so every harness can hold results side by side.
13
+ a directory missing either file is not discovered.
14
14
 
15
15
  ## task.md
16
16
 
@@ -27,9 +27,9 @@ Fix the failing check in the config below.
27
27
 
28
28
  each block is replaced in the prompt with a pointer ("Input file `config.json`
29
29
  is available in your working directory.") and written to disk. destinations
30
- must stay under the workdir, must not collide, and must not target `.claude/` or
31
- `.agents/`, since a fixture that writes agent config would be configuring its
32
- own examiner.
30
+ must stay under the workdir, must not collide, and must not target `.claude/`,
31
+ `.agents/`, or `.cursor/`, since a fixture that writes agent config would be
32
+ configuring its own examiner.
33
33
 
34
34
  write the task the way a user would write it. do not name the skill, describe
35
35
  its steps, or hint at the checklist: routing is part of what is being measured.
@@ -81,6 +81,10 @@ frontmatter block; body text mentioning the key does not count.
81
81
  ## the workdir
82
82
 
83
83
  per run, under `<root>/.skillcheck/scratch/<name>/`, rebuilt from scratch each
84
- time. the skill under test is installed at `.claude/skills/<skill>/` (and also
85
- `.agents/skills/<skill>/` on the codex harness) with its `evals/` directory
84
+ time. the skill under test is installed where the harness discovers skills —
85
+ `.claude/skills/<skill>/`, plus `.agents/skills/<skill>/` on codex, or
86
+ `.cursor/skills/<skill>/` alone on cursor — with its `evals/` directory
86
87
  excluded, so criteria never leak into the agent's context.
88
+
89
+ scenario quality is behavioral proof; [authoring](authoring.md) covers the
90
+ judgment layer lint and evals cannot grade.
package/docs/usage.md CHANGED
@@ -45,8 +45,16 @@ message, and writes no provenance sidecar. it is never reported as
45
45
  `FAIL score=0.0000`; only a real judged verdict can fail a run.
46
46
 
47
47
  defaults: `--harness claude`, agent `claude-opus-5`, judge `claude-opus-5`,
48
- `--max-turns 50`. on the codex harness, omitting `--agent` leaves the model to
49
- the Codex CLI's own default.
48
+ `--max-turns 50`. on the codex and cursor harnesses, omitting `--agent` leaves
49
+ the model to that CLI's own default.
50
+
51
+ `--harness cursor` drives the scenario through the Cursor Agent CLI
52
+ (`cursor-agent` on PATH) with the skill installed under `.cursor/skills/`;
53
+ `--agent` names a Cursor model id, e.g. `composer-2.5`. there is no promptfoo
54
+ cursor provider, so the run uses this package's own provider module, which
55
+ replays the CLI's `stream-json` output: the `result` event becomes the graded
56
+ output and `SKILL.md` reads under `.cursor/skills/` become the `skill-used`
57
+ evidence. the judge leg is unchanged.
50
58
 
51
59
  ## sweep
52
60
 
@@ -121,5 +129,6 @@ written inside the installed package.
121
129
  | `ANTHROPIC_API_KEY` | judge grades over `anthropic:messages:<model>` instead of the SDK |
122
130
  | `CODEX_HOME` (default `~/.codex`) | where the codex harness finds the local `codex` CLI login |
123
131
  | `OPENAI_API_KEY` | codex agent auth when there is no local login |
132
+ | `CURSOR_API_KEY` | cursor agent auth; a logged-in `cursor-agent` also works |
124
133
 
125
134
  the judge stays on the Anthropic selection regardless of the agent harness.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "0.2.0",
3
+ "version": "0.3.0",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {